diff --git a/CHANGELOG.md b/CHANGELOG.md index bf2aba46..caf84341 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,7 @@ - 对话中可隐藏不常用的共享专家(浏览器本地偏好)(#589) ### 修复 +- 远程 OCR 返回「未收到图片 / 请上传图片」等拒绝提示时不再当作正文入库(此前会被向量化、文档仍标记 ready),现按无可提取文本处理 - 环境变量值含换行(如粘贴的多行密钥)时,保存后再读取会被截断并残留引号;现按引号跨行读回,保证「写入 → 读取」往返保真 - 定时任务与系统任务(TLS 自动续期、自动备份)此前按宿主系统时区触发,现按 `default_timezone` 配置时区执行 diff --git a/src/octop/infra/knowledge/ocr.py b/src/octop/infra/knowledge/ocr.py index f1ba8c49..ff7681a1 100644 --- a/src/octop/infra/knowledge/ocr.py +++ b/src/octop/infra/knowledge/ocr.py @@ -5,6 +5,8 @@ from __future__ import annotations import asyncio import base64 import json +import logging +import re import threading from collections.abc import Iterator from dataclasses import dataclass @@ -18,6 +20,8 @@ from octop.infra.agents.providers.probe import build_probe_chat_model from octop.infra.utils.llm_text import llm_text_content from octop.infra.utils.runtime_packages import PackageInstallSpec, install_packages +logger = logging.getLogger(__name__) + OCR_IMAGE_SUFFIXES = frozenset({".png", ".jpg", ".jpeg", ".webp"}) _ENABLED_KEY = "knowledge_ocr_enabled" @@ -32,6 +36,32 @@ _OCR_PROMPT = ( "lists, and table rows. Return only the transcription, without commentary." ) +# A remote vision model that never received the image answers with a refusal such as +# "No image was attached. Please upload an image.". That is not source text: indexing it +# poisons the knowledge base and still marks the document ready. Refusals are short and +# addressed to the caller, so require both a short answer and a refusal phrasing; a real +# page that merely mentions uploading an image stays well above the length bound. +_OCR_REFUSAL_MAX_CHARS = 400 +_OCR_REFUSAL_RE = re.compile( + r"no image (?:was |is |has been )?(?:attached|provided|found|received|included)" + r"|(?:don'?t|do not|can'?t|cannot|unable to|didn'?t) " + r"(?:see|find|detect|receive|locate)[^.\n]{0,30}(?:image|picture|photo|scan|attachment)" + r"|please (?:attach|upload|provide|send|share)[^.\n]{0,40}" + r"(?:image|picture|photo|scan|attachment|file)" + r"|未(?:收到|看到|检测到|获取到)[^。\n]{0,10}(?:图片|图像|照片|附件)" + r"|请(?:上传|提供|重新上传|发送)[^。\n]{0,10}(?:图片|图像|照片|附件|文件)", + re.IGNORECASE, +) + + +def _is_no_image_refusal(text: str) -> bool: + """Return ``True`` when OCR output is the model asking for an image, not a transcription.""" + stripped = text.strip() + if not stripped or len(stripped) > _OCR_REFUSAL_MAX_CHARS: + return False + return _OCR_REFUSAL_RE.search(stripped) is not None + + _local_engine: Any | None = None _local_engine_lock = threading.Lock() @@ -297,6 +327,15 @@ class _RemoteOcr: ] ) text = llm_text_content(self._model.invoke([message])) - if text: - parts.append(text) + if not text: + continue + if _is_no_image_refusal(text): + # Indexing a refusal would mark the document ready with junk chunks; an empty + # result instead fails the document with "no extractable text". + logger.warning( + "remote OCR returned a no-image refusal for %s; treating the page as empty", + path.name, + ) + continue + parts.append(text) return "\n\n".join(parts) diff --git a/tests/unit/knowledge/test_ocr.py b/tests/unit/knowledge/test_ocr.py index 77abd692..ed3ee426 100644 --- a/tests/unit/knowledge/test_ocr.py +++ b/tests/unit/knowledge/test_ocr.py @@ -154,3 +154,70 @@ def test_remote_ocr_sends_image_block(tmp_path: Path, monkeypatch: pytest.Monkey content = messages[0].content assert content[1]["type"] == "image_url" assert content[1]["image_url"]["url"].startswith("data:image/png;base64,") + + +def _remote_extractor(monkeypatch: pytest.MonkeyPatch, reply: object) -> ocr._RemoteOcr: + """Build a ``_RemoteOcr`` whose model returns *reply* (str or a per-call sequence).""" + + class Model: + @staticmethod + def invoke(_value: list[object]) -> SimpleNamespace: + if isinstance(reply, list): + return SimpleNamespace(content=reply.pop(0)) + return SimpleNamespace(content=reply) + + monkeypatch.setattr(ocr, "build_probe_chat_model", lambda *_a, **_k: Model()) + return ocr._RemoteOcr(SimpleNamespace(), "vision") + + +@pytest.mark.parametrize( + "refusal", + [ + "No image was attached. Please upload an image.", + "I don't see an image attached to your message. Please upload the image you'd " + "like me to transcribe, and I'll provide the exact transcription.", + "未收到图片,请上传图片后重试。", + ], +) +def test_remote_ocr_ignores_no_image_refusal( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, refusal: str +) -> None: + """A refusal is not source text: indexing it would mark the document ready with junk.""" + image = tmp_path / "scan.png" + image.write_bytes(b"image") + + assert _remote_extractor(monkeypatch, refusal)(image) == "" + + +def test_remote_ocr_keeps_long_page_that_mentions_uploading_an_image( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The refusal filter must not drop real transcriptions that happen to mention an image.""" + image = tmp_path / "scan.png" + image.write_bytes(b"image") + page = ( + "No image was attached. Please upload an image. " + + "发票明细:办公用品 128.00 元,差旅费 340.00 元。" * 12 + ) + + assert _remote_extractor(monkeypatch, page)(image) == page + + +def test_remote_ocr_skips_refusal_page_but_keeps_transcribed_page( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """Refusals are dropped per page, so a partly readable document keeps its good pages.""" + pdf = tmp_path / "scan.pdf" + pdf.write_bytes(b"%PDF-1.4") + monkeypatch.setattr( + ocr, + "_image_inputs", + lambda _path: iter([(b"page-1", "image/png"), (b"page-2", "image/png")]), + ) + + extractor = _remote_extractor( + monkeypatch, + ["No image was attached. Please upload an image.", "第二页正文"], + ) + + assert extractor(pdf) == "第二页正文"