mirror of
https://github.com/debpalash/VoiceStudio.git
synced 2026-10-02 01:26:35 +08:00
Merge PR #2425 (strip prefilled reasoning)
This commit is contained in:
@@ -705,7 +705,9 @@ async def dub_translate(req: TranslateRequest):
|
||||
{"role": "user", "content": seg.text},
|
||||
],
|
||||
)
|
||||
out_text = (res.choices[0].message.content or "").strip()
|
||||
from services.llm_backend import _strip_reasoning
|
||||
out_text = _strip_reasoning(res.choices[0].message.content or "",
|
||||
prompt=f"{sys_for_attempt}\n{seg.text}")
|
||||
if not out_text:
|
||||
last_err = "empty LLM response"
|
||||
continue
|
||||
|
||||
@@ -234,7 +234,8 @@ def auto_extract(project_id: str, req: AutoExtractRequest):
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
)
|
||||
body = (res.choices[0].message.content or "").strip()
|
||||
from services.llm_backend import _strip_reasoning
|
||||
body = _strip_reasoning(res.choices[0].message.content or "", prompt=f"{system}\n{user}")
|
||||
except Exception as e:
|
||||
logger.warning("auto-extract LLM call failed: %s", e)
|
||||
# Scrub the provider error — some OpenAI-compatible providers echo the
|
||||
|
||||
@@ -42,9 +42,22 @@ _THINK_TAG_RE = re.compile(r"^\s*<(think|thinking|reasoning)>.*?</\1>", re.DOTAL
|
||||
#: Output truncated mid-thought (token cap, or a stop that never came) leaves
|
||||
#: the block unclosed. Still not an answer: drop from the tag to the end.
|
||||
_OPEN_THINK_RE = re.compile(r"^\s*<(think|thinking|reasoning)>.*\Z", re.DOTALL | re.IGNORECASE)
|
||||
#: Some chat templates open the block themselves (Spark-X2.5, Qwen3 thinking
|
||||
#: variants, DeepSeek-R1-0528 put ``<think>`` in the generation prompt), so a
|
||||
#: server without a reasoning parser returns only ``…reasoning</think>answer``.
|
||||
#: A closing tag with no opening tag before it marks the end of that block,
|
||||
#: unless the prompt itself contains the tag (see _strip_reasoning).
|
||||
_PREFILLED_THINK_RE = re.compile(
|
||||
r"^(?:(?!<(?:think|thinking|reasoning)>).)*?</(think|thinking|reasoning)>",
|
||||
re.DOTALL | re.IGNORECASE,
|
||||
)
|
||||
|
||||
|
||||
def _strip_reasoning(raw: str) -> str:
|
||||
def _prompt_text(messages: list[dict]) -> str:
|
||||
return "\n".join(m["content"] for m in messages if isinstance(m.get("content"), str))
|
||||
|
||||
|
||||
def _strip_reasoning(raw: str, prompt: str = "") -> str:
|
||||
"""Return the answer with any reasoning block removed.
|
||||
|
||||
Returns ``""`` when the response was *only* reasoning. That is deliberate:
|
||||
@@ -52,7 +65,14 @@ def _strip_reasoning(raw: str) -> str:
|
||||
input (dictation keeps the raw transcript, translation keeps the source),
|
||||
whereas handing back the model's private monologue as if it were the answer
|
||||
would silently overwrite the user's words with it.
|
||||
|
||||
``prompt`` is the text the reply answers. A bare closing tag only ends a
|
||||
prefilled block when the prompt does not contain that tag: an answer that
|
||||
translates or quotes input with a literal ``</think>`` must keep it.
|
||||
"""
|
||||
prefilled = _PREFILLED_THINK_RE.match(raw)
|
||||
if prefilled and f"</{prefilled.group(1)}>".lower() not in prompt.lower():
|
||||
raw = raw[prefilled.end():]
|
||||
while match := _THINK_TAG_RE.match(raw):
|
||||
raw = raw[match.end():]
|
||||
cleaned = _OPEN_THINK_RE.sub("", raw)
|
||||
@@ -285,7 +305,8 @@ class OpenAICompatBackend(LLMBackend):
|
||||
kw.pop("reasoning_effort", None)
|
||||
res = _create(**kw)
|
||||
|
||||
return _strip_reasoning(res.choices[0].message.content or "")
|
||||
return _strip_reasoning(res.choices[0].message.content or "",
|
||||
prompt=_prompt_text(messages))
|
||||
|
||||
def chat_messages_stream(self, *, messages: list[dict], timeout: Optional[float] = None,
|
||||
temperature: Optional[float] = None) -> Iterator[str]:
|
||||
|
||||
@@ -81,7 +81,8 @@ def _chat(client, model: str, timeout: float, *, system: str, user: str) -> str:
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
)
|
||||
return (res.choices[0].message.content or "").strip()
|
||||
from services.llm_backend import _strip_reasoning
|
||||
return _strip_reasoning(res.choices[0].message.content or "", prompt=f"{system}\n{user}")
|
||||
|
||||
|
||||
# ── Stage 1: auto-glossary (theme + terminology) ────────────────────────────
|
||||
|
||||
@@ -320,7 +320,9 @@ def _chat(client, *, system: str, user: str) -> str:
|
||||
{"role": "user", "content": user},
|
||||
],
|
||||
)
|
||||
return (res.choices[0].message.content or "").strip()
|
||||
from services.llm_backend import _strip_reasoning
|
||||
return _strip_reasoning(res.choices[0].message.content or "",
|
||||
prompt=f"{system}\n{user}")
|
||||
except Exception as e: # noqa: BLE001 — re-raised unless a retryable 429
|
||||
wait = _retry_after_seconds(e)
|
||||
if wait is None or attempts >= 1:
|
||||
|
||||
@@ -396,3 +396,26 @@ def test_internal_type_error_is_not_retried_or_memoized(monkeypatch):
|
||||
def test_multiple_leading_reasoning_blocks_keep_the_final_answer():
|
||||
from services.llm_backend import _strip_reasoning
|
||||
assert _strip_reasoning('<think>one</think>\n<thinking>two</thinking>Answer.') == 'Answer.'
|
||||
|
||||
|
||||
def test_prefilled_reasoning_block_is_stripped():
|
||||
"""Templates that put <think> in the prompt (Spark-X2.5, Qwen3 thinking
|
||||
variants) leave only the closing tag in the reply when the server runs
|
||||
without a reasoning parser."""
|
||||
from services.llm_backend import _strip_reasoning
|
||||
assert _strip_reasoning("The user wants Spanish.\n</think>\n\nHola.") == "Hola."
|
||||
assert _strip_reasoning("weighing it</thinking>Answer.") == "Answer."
|
||||
assert _strip_reasoning("only reasoning, then</think>") == ""
|
||||
# An answer that merely mentions a tag is still left alone.
|
||||
assert _strip_reasoning("Close it with <think>x</think> here.") == "Close it with <think>x</think> here."
|
||||
|
||||
|
||||
def test_literal_closing_tag_from_the_prompt_is_kept():
|
||||
"""A reply may repeat a bare </think> only because the input had one —
|
||||
translating that input must not cut the answer at the tag."""
|
||||
from services.llm_backend import _strip_reasoning
|
||||
line = "Use </think> to close the block."
|
||||
assert _strip_reasoning(line, prompt="Translate to Spanish:\n" + line) == line
|
||||
assert _strip_reasoning("Usa </THINK> para cerrar.", prompt=line) == "Usa </THINK> para cerrar."
|
||||
# A tag the prompt never contained still ends a prefilled block.
|
||||
assert _strip_reasoning("weighing it</think>Answer.", prompt="Translate: hi") == "Answer."
|
||||
|
||||
@@ -483,3 +483,34 @@ async def test_custom_style_reaches_direct_translation(monkeypatch):
|
||||
direct = [c for c in client.calls if _is_direct_call(client, c)]
|
||||
assert direct
|
||||
assert all("Keep it warm and conversational; preserve jokes." in client.system_of(c) for c in direct)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_reasoning_model_output_never_reaches_the_dub(monkeypatch):
|
||||
"""Spark-X2.5 and Qwen3 thinking templates prefill <think>, so every reply
|
||||
from a server without a reasoning parser starts with the monologue and a
|
||||
bare </think>. None of it may land in the literal, the critique or the
|
||||
final line."""
|
||||
from api.routers import dub_translate
|
||||
|
||||
def script(kw):
|
||||
sys_msg = kw["messages"][0]["content"]
|
||||
if "script reviewer" in sys_msg:
|
||||
answer = "Too stiff for spoken dialogue."
|
||||
elif "script writer" in sys_msg:
|
||||
answer = "enciende la parrilla ya"
|
||||
else:
|
||||
answer = "procede a encender la parrilla ahora"
|
||||
return "The user wants Spanish, keep it natural.\n</think>\n\n" + answer
|
||||
|
||||
client = _ScriptedLLMClient(script)
|
||||
_wire_skill_client(monkeypatch, client)
|
||||
|
||||
resp = await dub_translate.dub_translate(
|
||||
_req(_segs("Fire up the grill now."), auto_glossary=False))
|
||||
row = resp["translated"][0]
|
||||
assert row["text"] == "enciende la parrilla ya"
|
||||
assert row["literal"] == "procede a encender la parrilla ahora"
|
||||
polish = [c for c in client.calls if _is_polish_call(client, c)][0]
|
||||
assert "</think>" not in client.user_of(polish)
|
||||
assert "The user wants" not in client.user_of(polish)
|
||||
|
||||
@@ -72,3 +72,33 @@ def test_auto_extract_scrubs_provider_error(monkeypatch):
|
||||
assert secret not in detail
|
||||
assert home not in detail
|
||||
assert "***REDACTED***" in detail
|
||||
|
||||
|
||||
def test_auto_extract_ignores_terms_inside_reasoning(monkeypatch):
|
||||
"""A reasoning model drafts candidate pairs in its monologue; only the
|
||||
answer after </think> may become glossary rows."""
|
||||
from api.routers import glossary
|
||||
from services import llm_skills
|
||||
from core.db import ensure_schema
|
||||
|
||||
ensure_schema()
|
||||
body = (
|
||||
"Candidates:\nMarcus || WRONG || draft\n</think>\n"
|
||||
"Marcus || Marcus || character name\n"
|
||||
)
|
||||
|
||||
class _Completions:
|
||||
def create(self, **kw):
|
||||
msg = type("M", (), {"content": body})
|
||||
return type("R", (), {"choices": [type("C", (), {"message": msg})]})
|
||||
|
||||
class _Handle:
|
||||
client = type("Client", (), {"chat": type("Chat", (), {"completions": _Completions()})()})()
|
||||
model = "m"
|
||||
timeout = 1.0
|
||||
|
||||
monkeypatch.setattr(llm_skills, "resolve_skill_client", lambda sid: _Handle())
|
||||
|
||||
out = glossary.auto_extract("proj-reasoning", _req(target_lang="es", segments=[{"text": "Hello Marcus"}]))
|
||||
assert out["proposed"] == 1
|
||||
assert [t["target"] for t in out["terms"] if t["source"] == "Marcus"] == ["Marcus"]
|
||||
|
||||
@@ -381,3 +381,22 @@ def test_chat_non_429_does_not_retry(monkeypatch):
|
||||
tr._chat(client, system="s", user="u")
|
||||
assert client.chat.completions.create.call_count == 1
|
||||
assert not slept
|
||||
|
||||
|
||||
def test_chat_strips_prefilled_reasoning(monkeypatch):
|
||||
"""A local reasoning model served without a reasoning parser must not leak
|
||||
its monologue into the Cinematic output."""
|
||||
client = MagicMock()
|
||||
client.chat.completions.create.return_value = MagicMock(
|
||||
choices=[MagicMock(message=MagicMock(content="Keep it short.\n</think>\n\nHola, amigo."))])
|
||||
monkeypatch.setattr(tr, "_llm_model", lambda: "test-model")
|
||||
assert tr._chat(client, system="s", user="u") == "Hola, amigo."
|
||||
|
||||
|
||||
def test_chat_keeps_a_closing_tag_quoted_from_the_source(monkeypatch):
|
||||
client = MagicMock()
|
||||
client.chat.completions.create.return_value = MagicMock(
|
||||
choices=[MagicMock(message=MagicMock(content="Usa </think> para cerrar el bloque."))])
|
||||
monkeypatch.setattr(tr, "_llm_model", lambda: "test-model")
|
||||
out = tr._chat(client, system="s", user="Use </think> to close the block.")
|
||||
assert out == "Usa </think> para cerrar el bloque."
|
||||
|
||||
Reference in New Issue
Block a user