mirror of
https://github.com/TencentCloud/Octop.git
synced 2026-10-02 07:34:38 +08:00
@@ -12,6 +12,7 @@
|
||||
- 对话中可隐藏不常用的共享专家(浏览器本地偏好)(#589)
|
||||
|
||||
### 修复
|
||||
- 远程 OCR 返回「未收到图片 / 请上传图片」等拒绝提示时不再当作正文入库(此前会被向量化、文档仍标记 ready),现按无可提取文本处理
|
||||
- 环境变量值含换行(如粘贴的多行密钥)时,保存后再读取会被截断并残留引号;现按引号跨行读回,保证「写入 → 读取」往返保真
|
||||
- 定时任务与系统任务(TLS 自动续期、自动备份)此前按宿主系统时区触发,现按 `default_timezone` 配置时区执行
|
||||
|
||||
|
||||
@@ -5,6 +5,8 @@ from __future__ import annotations
|
||||
import asyncio
|
||||
import base64
|
||||
import json
|
||||
import logging
|
||||
import re
|
||||
import threading
|
||||
from collections.abc import Iterator
|
||||
from dataclasses import dataclass
|
||||
@@ -18,6 +20,8 @@ from octop.infra.agents.providers.probe import build_probe_chat_model
|
||||
from octop.infra.utils.llm_text import llm_text_content
|
||||
from octop.infra.utils.runtime_packages import PackageInstallSpec, install_packages
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
OCR_IMAGE_SUFFIXES = frozenset({".png", ".jpg", ".jpeg", ".webp"})
|
||||
|
||||
_ENABLED_KEY = "knowledge_ocr_enabled"
|
||||
@@ -32,6 +36,32 @@ _OCR_PROMPT = (
|
||||
"lists, and table rows. Return only the transcription, without commentary."
|
||||
)
|
||||
|
||||
# A remote vision model that never received the image answers with a refusal such as
|
||||
# "No image was attached. Please upload an image.". That is not source text: indexing it
|
||||
# poisons the knowledge base and still marks the document ready. Refusals are short and
|
||||
# addressed to the caller, so require both a short answer and a refusal phrasing; a real
|
||||
# page that merely mentions uploading an image stays well above the length bound.
|
||||
_OCR_REFUSAL_MAX_CHARS = 400
|
||||
_OCR_REFUSAL_RE = re.compile(
|
||||
r"no image (?:was |is |has been )?(?:attached|provided|found|received|included)"
|
||||
r"|(?:don'?t|do not|can'?t|cannot|unable to|didn'?t) "
|
||||
r"(?:see|find|detect|receive|locate)[^.\n]{0,30}(?:image|picture|photo|scan|attachment)"
|
||||
r"|please (?:attach|upload|provide|send|share)[^.\n]{0,40}"
|
||||
r"(?:image|picture|photo|scan|attachment|file)"
|
||||
r"|未(?:收到|看到|检测到|获取到)[^。\n]{0,10}(?:图片|图像|照片|附件)"
|
||||
r"|请(?:上传|提供|重新上传|发送)[^。\n]{0,10}(?:图片|图像|照片|附件|文件)",
|
||||
re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _is_no_image_refusal(text: str) -> bool:
|
||||
"""Return ``True`` when OCR output is the model asking for an image, not a transcription."""
|
||||
stripped = text.strip()
|
||||
if not stripped or len(stripped) > _OCR_REFUSAL_MAX_CHARS:
|
||||
return False
|
||||
return _OCR_REFUSAL_RE.search(stripped) is not None
|
||||
|
||||
|
||||
_local_engine: Any | None = None
|
||||
_local_engine_lock = threading.Lock()
|
||||
|
||||
@@ -297,6 +327,15 @@ class _RemoteOcr:
|
||||
]
|
||||
)
|
||||
text = llm_text_content(self._model.invoke([message]))
|
||||
if text:
|
||||
parts.append(text)
|
||||
if not text:
|
||||
continue
|
||||
if _is_no_image_refusal(text):
|
||||
# Indexing a refusal would mark the document ready with junk chunks; an empty
|
||||
# result instead fails the document with "no extractable text".
|
||||
logger.warning(
|
||||
"remote OCR returned a no-image refusal for %s; treating the page as empty",
|
||||
path.name,
|
||||
)
|
||||
continue
|
||||
parts.append(text)
|
||||
return "\n\n".join(parts)
|
||||
|
||||
@@ -154,3 +154,70 @@ def test_remote_ocr_sends_image_block(tmp_path: Path, monkeypatch: pytest.Monkey
|
||||
content = messages[0].content
|
||||
assert content[1]["type"] == "image_url"
|
||||
assert content[1]["image_url"]["url"].startswith("data:image/png;base64,")
|
||||
|
||||
|
||||
def _remote_extractor(monkeypatch: pytest.MonkeyPatch, reply: object) -> ocr._RemoteOcr:
|
||||
"""Build a ``_RemoteOcr`` whose model returns *reply* (str or a per-call sequence)."""
|
||||
|
||||
class Model:
|
||||
@staticmethod
|
||||
def invoke(_value: list[object]) -> SimpleNamespace:
|
||||
if isinstance(reply, list):
|
||||
return SimpleNamespace(content=reply.pop(0))
|
||||
return SimpleNamespace(content=reply)
|
||||
|
||||
monkeypatch.setattr(ocr, "build_probe_chat_model", lambda *_a, **_k: Model())
|
||||
return ocr._RemoteOcr(SimpleNamespace(), "vision")
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"refusal",
|
||||
[
|
||||
"No image was attached. Please upload an image.",
|
||||
"I don't see an image attached to your message. Please upload the image you'd "
|
||||
"like me to transcribe, and I'll provide the exact transcription.",
|
||||
"未收到图片,请上传图片后重试。",
|
||||
],
|
||||
)
|
||||
def test_remote_ocr_ignores_no_image_refusal(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, refusal: str
|
||||
) -> None:
|
||||
"""A refusal is not source text: indexing it would mark the document ready with junk."""
|
||||
image = tmp_path / "scan.png"
|
||||
image.write_bytes(b"image")
|
||||
|
||||
assert _remote_extractor(monkeypatch, refusal)(image) == ""
|
||||
|
||||
|
||||
def test_remote_ocr_keeps_long_page_that_mentions_uploading_an_image(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""The refusal filter must not drop real transcriptions that happen to mention an image."""
|
||||
image = tmp_path / "scan.png"
|
||||
image.write_bytes(b"image")
|
||||
page = (
|
||||
"No image was attached. Please upload an image. "
|
||||
+ "发票明细:办公用品 128.00 元,差旅费 340.00 元。" * 12
|
||||
)
|
||||
|
||||
assert _remote_extractor(monkeypatch, page)(image) == page
|
||||
|
||||
|
||||
def test_remote_ocr_skips_refusal_page_but_keeps_transcribed_page(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
) -> None:
|
||||
"""Refusals are dropped per page, so a partly readable document keeps its good pages."""
|
||||
pdf = tmp_path / "scan.pdf"
|
||||
pdf.write_bytes(b"%PDF-1.4")
|
||||
monkeypatch.setattr(
|
||||
ocr,
|
||||
"_image_inputs",
|
||||
lambda _path: iter([(b"page-1", "image/png"), (b"page-2", "image/png")]),
|
||||
)
|
||||
|
||||
extractor = _remote_extractor(
|
||||
monkeypatch,
|
||||
["No image was attached. Please upload an image.", "第二页正文"],
|
||||
)
|
||||
|
||||
assert extractor(pdf) == "第二页正文"
|
||||
|
||||
Reference in New Issue
Block a user