fix(knowledge): 远程 OCR 的「请上传图片」拒绝提示不再作为正文入库

Merged from #867.
This commit is contained in:
jubaoliang
2026-09-21 03:02:35 +00:00
parent c7919ff22e
commit aba6b0bb8f
3 changed files with 109 additions and 2 deletions
+1
View File
@@ -12,6 +12,7 @@
- 对话中可隐藏不常用的共享专家(浏览器本地偏好)(#589)
### 修复
- 远程 OCR 返回「未收到图片 / 请上传图片」等拒绝提示时不再当作正文入库(此前会被向量化、文档仍标记 ready),现按无可提取文本处理
- 环境变量值含换行(如粘贴的多行密钥)时,保存后再读取会被截断并残留引号;现按引号跨行读回,保证「写入 → 读取」往返保真
- 定时任务与系统任务(TLS 自动续期、自动备份)此前按宿主系统时区触发,现按 `default_timezone` 配置时区执行
+41 -2
View File
@@ -5,6 +5,8 @@ from __future__ import annotations
import asyncio
import base64
import json
import logging
import re
import threading
from collections.abc import Iterator
from dataclasses import dataclass
@@ -18,6 +20,8 @@ from octop.infra.agents.providers.probe import build_probe_chat_model
from octop.infra.utils.llm_text import llm_text_content
from octop.infra.utils.runtime_packages import PackageInstallSpec, install_packages
logger = logging.getLogger(__name__)
OCR_IMAGE_SUFFIXES = frozenset({".png", ".jpg", ".jpeg", ".webp"})
_ENABLED_KEY = "knowledge_ocr_enabled"
@@ -32,6 +36,32 @@ _OCR_PROMPT = (
"lists, and table rows. Return only the transcription, without commentary."
)
# A remote vision model that never received the image answers with a refusal such as
# "No image was attached. Please upload an image.". That is not source text: indexing it
# poisons the knowledge base and still marks the document ready. Refusals are short and
# addressed to the caller, so require both a short answer and a refusal phrasing; a real
# page that merely mentions uploading an image stays well above the length bound.
_OCR_REFUSAL_MAX_CHARS = 400
_OCR_REFUSAL_RE = re.compile(
r"no image (?:was |is |has been )?(?:attached|provided|found|received|included)"
r"|(?:don'?t|do not|can'?t|cannot|unable to|didn'?t) "
r"(?:see|find|detect|receive|locate)[^.\n]{0,30}(?:image|picture|photo|scan|attachment)"
r"|please (?:attach|upload|provide|send|share)[^.\n]{0,40}"
r"(?:image|picture|photo|scan|attachment|file)"
r"|未(?:收到|看到|检测到|获取到)[^。\n]{0,10}(?:图片|图像|照片|附件)"
r"|请(?:上传|提供|重新上传|发送)[^。\n]{0,10}(?:图片|图像|照片|附件|文件)",
re.IGNORECASE,
)
def _is_no_image_refusal(text: str) -> bool:
"""Return ``True`` when OCR output is the model asking for an image, not a transcription."""
stripped = text.strip()
if not stripped or len(stripped) > _OCR_REFUSAL_MAX_CHARS:
return False
return _OCR_REFUSAL_RE.search(stripped) is not None
_local_engine: Any | None = None
_local_engine_lock = threading.Lock()
@@ -297,6 +327,15 @@ class _RemoteOcr:
]
)
text = llm_text_content(self._model.invoke([message]))
if text:
parts.append(text)
if not text:
continue
if _is_no_image_refusal(text):
# Indexing a refusal would mark the document ready with junk chunks; an empty
# result instead fails the document with "no extractable text".
logger.warning(
"remote OCR returned a no-image refusal for %s; treating the page as empty",
path.name,
)
continue
parts.append(text)
return "\n\n".join(parts)
+67
View File
@@ -154,3 +154,70 @@ def test_remote_ocr_sends_image_block(tmp_path: Path, monkeypatch: pytest.Monkey
content = messages[0].content
assert content[1]["type"] == "image_url"
assert content[1]["image_url"]["url"].startswith("data:image/png;base64,")
def _remote_extractor(monkeypatch: pytest.MonkeyPatch, reply: object) -> ocr._RemoteOcr:
"""Build a ``_RemoteOcr`` whose model returns *reply* (str or a per-call sequence)."""
class Model:
@staticmethod
def invoke(_value: list[object]) -> SimpleNamespace:
if isinstance(reply, list):
return SimpleNamespace(content=reply.pop(0))
return SimpleNamespace(content=reply)
monkeypatch.setattr(ocr, "build_probe_chat_model", lambda *_a, **_k: Model())
return ocr._RemoteOcr(SimpleNamespace(), "vision")
@pytest.mark.parametrize(
"refusal",
[
"No image was attached. Please upload an image.",
"I don't see an image attached to your message. Please upload the image you'd "
"like me to transcribe, and I'll provide the exact transcription.",
"未收到图片,请上传图片后重试。",
],
)
def test_remote_ocr_ignores_no_image_refusal(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, refusal: str
) -> None:
"""A refusal is not source text: indexing it would mark the document ready with junk."""
image = tmp_path / "scan.png"
image.write_bytes(b"image")
assert _remote_extractor(monkeypatch, refusal)(image) == ""
def test_remote_ocr_keeps_long_page_that_mentions_uploading_an_image(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""The refusal filter must not drop real transcriptions that happen to mention an image."""
image = tmp_path / "scan.png"
image.write_bytes(b"image")
page = (
"No image was attached. Please upload an image. "
+ "发票明细:办公用品 128.00 元,差旅费 340.00 元。" * 12
)
assert _remote_extractor(monkeypatch, page)(image) == page
def test_remote_ocr_skips_refusal_page_but_keeps_transcribed_page(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
"""Refusals are dropped per page, so a partly readable document keeps its good pages."""
pdf = tmp_path / "scan.pdf"
pdf.write_bytes(b"%PDF-1.4")
monkeypatch.setattr(
ocr,
"_image_inputs",
lambda _path: iter([(b"page-1", "image/png"), (b"page-2", "image/png")]),
)
extractor = _remote_extractor(
monkeypatch,
["No image was attached. Please upload an image.", "第二页正文"],
)
assert extractor(pdf) == "第二页正文"