Merge PR #2425 (strip prefilled reasoning)

This commit is contained in:
Palash Debnath
2026-10-01 18:45:47 +05:30
9 changed files with 136 additions and 6 deletions
+3 -1
View File
@@ -705,7 +705,9 @@ async def dub_translate(req: TranslateRequest):
{"role": "user", "content": seg.text},
],
)
out_text = (res.choices[0].message.content or "").strip()
from services.llm_backend import _strip_reasoning
out_text = _strip_reasoning(res.choices[0].message.content or "",
prompt=f"{sys_for_attempt}\n{seg.text}")
if not out_text:
last_err = "empty LLM response"
continue
+2 -1
View File
@@ -234,7 +234,8 @@ def auto_extract(project_id: str, req: AutoExtractRequest):
{"role": "user", "content": user},
],
)
body = (res.choices[0].message.content or "").strip()
from services.llm_backend import _strip_reasoning
body = _strip_reasoning(res.choices[0].message.content or "", prompt=f"{system}\n{user}")
except Exception as e:
logger.warning("auto-extract LLM call failed: %s", e)
# Scrub the provider error — some OpenAI-compatible providers echo the
+23 -2
View File
@@ -42,9 +42,22 @@ _THINK_TAG_RE = re.compile(r"^\s*<(think|thinking|reasoning)>.*?</\1>", re.DOTAL
#: Output truncated mid-thought (token cap, or a stop that never came) leaves
#: the block unclosed. Still not an answer: drop from the tag to the end.
_OPEN_THINK_RE = re.compile(r"^\s*<(think|thinking|reasoning)>.*\Z", re.DOTALL | re.IGNORECASE)
#: Some chat templates open the block themselves (Spark-X2.5, Qwen3 thinking
#: variants, DeepSeek-R1-0528 put ``<think>`` in the generation prompt), so a
#: server without a reasoning parser returns only ``…reasoning</think>answer``.
#: A closing tag with no opening tag before it marks the end of that block,
#: unless the prompt itself contains the tag (see _strip_reasoning).
_PREFILLED_THINK_RE = re.compile(
r"^(?:(?!<(?:think|thinking|reasoning)>).)*?</(think|thinking|reasoning)>",
re.DOTALL | re.IGNORECASE,
)
def _strip_reasoning(raw: str) -> str:
def _prompt_text(messages: list[dict]) -> str:
return "\n".join(m["content"] for m in messages if isinstance(m.get("content"), str))
def _strip_reasoning(raw: str, prompt: str = "") -> str:
"""Return the answer with any reasoning block removed.
Returns ``""`` when the response was *only* reasoning. That is deliberate:
@@ -52,7 +65,14 @@ def _strip_reasoning(raw: str) -> str:
input (dictation keeps the raw transcript, translation keeps the source),
whereas handing back the model's private monologue as if it were the answer
would silently overwrite the user's words with it.
``prompt`` is the text the reply answers. A bare closing tag only ends a
prefilled block when the prompt does not contain that tag: an answer that
translates or quotes input with a literal ``</think>`` must keep it.
"""
prefilled = _PREFILLED_THINK_RE.match(raw)
if prefilled and f"</{prefilled.group(1)}>".lower() not in prompt.lower():
raw = raw[prefilled.end():]
while match := _THINK_TAG_RE.match(raw):
raw = raw[match.end():]
cleaned = _OPEN_THINK_RE.sub("", raw)
@@ -285,7 +305,8 @@ class OpenAICompatBackend(LLMBackend):
kw.pop("reasoning_effort", None)
res = _create(**kw)
return _strip_reasoning(res.choices[0].message.content or "")
return _strip_reasoning(res.choices[0].message.content or "",
prompt=_prompt_text(messages))
def chat_messages_stream(self, *, messages: list[dict], timeout: Optional[float] = None,
temperature: Optional[float] = None) -> Iterator[str]:
+2 -1
View File
@@ -81,7 +81,8 @@ def _chat(client, model: str, timeout: float, *, system: str, user: str) -> str:
{"role": "user", "content": user},
],
)
return (res.choices[0].message.content or "").strip()
from services.llm_backend import _strip_reasoning
return _strip_reasoning(res.choices[0].message.content or "", prompt=f"{system}\n{user}")
# ── Stage 1: auto-glossary (theme + terminology) ────────────────────────────
+3 -1
View File
@@ -320,7 +320,9 @@ def _chat(client, *, system: str, user: str) -> str:
{"role": "user", "content": user},
],
)
return (res.choices[0].message.content or "").strip()
from services.llm_backend import _strip_reasoning
return _strip_reasoning(res.choices[0].message.content or "",
prompt=f"{system}\n{user}")
except Exception as e: # noqa: BLE001 — re-raised unless a retryable 429
wait = _retry_after_seconds(e)
if wait is None or attempts >= 1:
@@ -396,3 +396,26 @@ def test_internal_type_error_is_not_retried_or_memoized(monkeypatch):
def test_multiple_leading_reasoning_blocks_keep_the_final_answer():
from services.llm_backend import _strip_reasoning
assert _strip_reasoning('<think>one</think>\n<thinking>two</thinking>Answer.') == 'Answer.'
def test_prefilled_reasoning_block_is_stripped():
"""Templates that put <think> in the prompt (Spark-X2.5, Qwen3 thinking
variants) leave only the closing tag in the reply when the server runs
without a reasoning parser."""
from services.llm_backend import _strip_reasoning
assert _strip_reasoning("The user wants Spanish.\n</think>\n\nHola.") == "Hola."
assert _strip_reasoning("weighing it</thinking>Answer.") == "Answer."
assert _strip_reasoning("only reasoning, then</think>") == ""
# An answer that merely mentions a tag is still left alone.
assert _strip_reasoning("Close it with <think>x</think> here.") == "Close it with <think>x</think> here."
def test_literal_closing_tag_from_the_prompt_is_kept():
"""A reply may repeat a bare </think> only because the input had one —
translating that input must not cut the answer at the tag."""
from services.llm_backend import _strip_reasoning
line = "Use </think> to close the block."
assert _strip_reasoning(line, prompt="Translate to Spanish:\n" + line) == line
assert _strip_reasoning("Usa </THINK> para cerrar.", prompt=line) == "Usa </THINK> para cerrar."
# A tag the prompt never contained still ends a prefilled block.
assert _strip_reasoning("weighing it</think>Answer.", prompt="Translate: hi") == "Answer."
+31
View File
@@ -483,3 +483,34 @@ async def test_custom_style_reaches_direct_translation(monkeypatch):
direct = [c for c in client.calls if _is_direct_call(client, c)]
assert direct
assert all("Keep it warm and conversational; preserve jokes." in client.system_of(c) for c in direct)
@pytest.mark.asyncio
async def test_reasoning_model_output_never_reaches_the_dub(monkeypatch):
"""Spark-X2.5 and Qwen3 thinking templates prefill <think>, so every reply
from a server without a reasoning parser starts with the monologue and a
bare </think>. None of it may land in the literal, the critique or the
final line."""
from api.routers import dub_translate
def script(kw):
sys_msg = kw["messages"][0]["content"]
if "script reviewer" in sys_msg:
answer = "Too stiff for spoken dialogue."
elif "script writer" in sys_msg:
answer = "enciende la parrilla ya"
else:
answer = "procede a encender la parrilla ahora"
return "The user wants Spanish, keep it natural.\n</think>\n\n" + answer
client = _ScriptedLLMClient(script)
_wire_skill_client(monkeypatch, client)
resp = await dub_translate.dub_translate(
_req(_segs("Fire up the grill now."), auto_glossary=False))
row = resp["translated"][0]
assert row["text"] == "enciende la parrilla ya"
assert row["literal"] == "procede a encender la parrilla ahora"
polish = [c for c in client.calls if _is_polish_call(client, c)][0]
assert "</think>" not in client.user_of(polish)
assert "The user wants" not in client.user_of(polish)
+30
View File
@@ -72,3 +72,33 @@ def test_auto_extract_scrubs_provider_error(monkeypatch):
assert secret not in detail
assert home not in detail
assert "***REDACTED***" in detail
def test_auto_extract_ignores_terms_inside_reasoning(monkeypatch):
"""A reasoning model drafts candidate pairs in its monologue; only the
answer after </think> may become glossary rows."""
from api.routers import glossary
from services import llm_skills
from core.db import ensure_schema
ensure_schema()
body = (
"Candidates:\nMarcus || WRONG || draft\n</think>\n"
"Marcus || Marcus || character name\n"
)
class _Completions:
def create(self, **kw):
msg = type("M", (), {"content": body})
return type("R", (), {"choices": [type("C", (), {"message": msg})]})
class _Handle:
client = type("Client", (), {"chat": type("Chat", (), {"completions": _Completions()})()})()
model = "m"
timeout = 1.0
monkeypatch.setattr(llm_skills, "resolve_skill_client", lambda sid: _Handle())
out = glossary.auto_extract("proj-reasoning", _req(target_lang="es", segments=[{"text": "Hello Marcus"}]))
assert out["proposed"] == 1
assert [t["target"] for t in out["terms"] if t["source"] == "Marcus"] == ["Marcus"]
+19
View File
@@ -381,3 +381,22 @@ def test_chat_non_429_does_not_retry(monkeypatch):
tr._chat(client, system="s", user="u")
assert client.chat.completions.create.call_count == 1
assert not slept
def test_chat_strips_prefilled_reasoning(monkeypatch):
"""A local reasoning model served without a reasoning parser must not leak
its monologue into the Cinematic output."""
client = MagicMock()
client.chat.completions.create.return_value = MagicMock(
choices=[MagicMock(message=MagicMock(content="Keep it short.\n</think>\n\nHola, amigo."))])
monkeypatch.setattr(tr, "_llm_model", lambda: "test-model")
assert tr._chat(client, system="s", user="u") == "Hola, amigo."
def test_chat_keeps_a_closing_tag_quoted_from_the_source(monkeypatch):
client = MagicMock()
client.chat.completions.create.return_value = MagicMock(
choices=[MagicMock(message=MagicMock(content="Usa </think> para cerrar el bloque."))])
monkeypatch.setattr(tr, "_llm_model", lambda: "test-model")
out = tr._chat(client, system="s", user="Use </think> to close the block.")
assert out == "Usa </think> para cerrar el bloque."