Merge commit 'refs/pull/2431/head' of github.com:debpalash/VoiceStudio into triage/clone-asr-llm

This commit is contained in:
Palash Debnath
2026-10-01 18:44:51 +05:30
4 changed files with 132 additions and 5 deletions
+1
View File
@@ -98,6 +98,7 @@ metadata and the backend fallback mirror it.
- MLX Qwen3-TTS receives the selected language correctly; MeloTTS explains missing text resources without downloading during generation (#2419)
- The language picker offers only the languages each MLX-Audio model supports (Kokoro, CSM, Qwen3-TTS, Dia, Chatterbox, MeloTTS, OuteTTS), per their model cards, instead of every language (#977)
- The call agent holds back reasoning from chat templates that prefill the opening tag instead of speaking it to the caller (#2428) — thanks @swadhinbiswas!
- The Enter that confirms Korean, Japanese or Chinese input no longer also submits project renames, language search, pronunciation, worker or MCP fields (#2338) — thanks @HEOJUNFO!
- English text normalization speaks a dollar amount followed by a period or comma ("It costs $5.") instead of leaving the digits (#2390) — thanks @kevin9327!
- Voice Design sends descriptions unchanged to free-text engines and restores the original design when reopening a take (#2401) — thanks @CauaMatheus and @dominikj-cf!
+40 -4
View File
@@ -190,6 +190,16 @@ def guard_sensitive(sentence: str, brief: str) -> str:
# ── Structured LLM reply ────────────────────────────────────────────────────
_THINK_OPEN_RE = re.compile(r"^\s*<(think|thinking|reasoning)>", re.IGNORECASE)
# A chat template that prefills the opening tag into the prompt leaves only
# the closing one on the wire (#2428) — there is no opening tag to match.
_THINK_CLOSE_RE = re.compile(r"</(?:think|thinking|reasoning)>", re.IGNORECASE)
# A line-start SAY: later in an undecided body marks where prefilled
# thinking ends, for a model that streams no closing tag either (#2428).
# Case-insensitive like _SAY_RE: a lowercase boundary must not leave the
# reasoning to finish()'s plain path. Only consulted at finish — a draft
# "SAY:" line *inside* reasoning must not switch to tagged while a
# closing tag may still arrive.
_SAY_LINE_RE = re.compile(r"(?m)^[ \t]*SAY\s*:", re.IGNORECASE)
_TAG_RE = re.compile(r"(?:^|\s)(ACTION|OUTCOME)\s*:", re.IGNORECASE)
_SAY_RE = re.compile(r"^\s*SAY\s*:\s*", re.IGNORECASE)
_ACTION_VALUE_RE = re.compile(r"ACTION\s*:\s*([A-Za-z_\- ]+)", re.IGNORECASE)
@@ -226,7 +236,7 @@ class ReplyParser:
self.action = "none"
self.outcome: str | None = None
def _body(self) -> str | None:
def _body(self, final: bool = False) -> str | None:
text = self.raw
match = _THINK_OPEN_RE.match(text)
if match:
@@ -234,6 +244,25 @@ class ReplyParser:
if not close:
return None
text = text[close.end():]
else:
# No opening tag: either there is no reasoning at all, or the
# chat template prefilled the opening tag into the prompt and
# the model streams only …SAY: (#2428).
close = _THINK_CLOSE_RE.search(text)
if close:
text = text[close.end():]
# A line-start SAY: marks where thinking ends for the model that
# never sent a closing tag either (#2428) — but only at finish, and
# only while no mode has been chosen (#2431 review): a draft "SAY:"
# line inside still-streaming reasoning must not flip the parser to
# tagged before a closing tag establishes the real boundary, and a
# body already streaming as tagged/json must never be re-sliced
# here, or finish() would discard text ``_emitted`` already counts
# (or, for json, drop a say that parses). While streaming everything
# without a format marker is held anyway, so waiting costs nothing.
say = _SAY_LINE_RE.search(text) if final and self.mode in (None, "plain") else None
if say:
text = text[say.start():]
return text.lstrip()
def _decide_mode(self, body: str, final: bool) -> None:
@@ -243,8 +272,15 @@ class ReplyParser:
self.mode = "json"
elif _SAY_RE.match(body):
self.mode = "tagged"
elif final or len(body) >= 4 or not "say:".startswith(body[:4].lower()):
elif final:
self.mode = "plain"
# Not final and neither format marker: hold. The system prompt
# requires replies to start with SAY: or {, so anything else may be
# prefilled thinking (#2428) — the first spoken sentence cannot be
# taken back, and this text reaches a third party on the phone.
# ``_body`` reveals the format when the closing tag or a later
# line-start SAY: arrives; ``finish`` settles a reply with neither
# as plain, as it always did.
def _say(self, body: str, final: bool) -> str:
if self.mode == "json":
@@ -269,7 +305,7 @@ class ReplyParser:
def feed(self, delta: str) -> str:
self.raw += delta or ""
body = self._body()
body = self._body(final=False)
if body is None:
return ""
self._decide_mode(body, final=False)
@@ -278,7 +314,7 @@ class ReplyParser:
return self._take(self._say(body, final=False))
def finish(self) -> str:
body = self._body()
body = self._body(final=True)
if body is None: # never left the reasoning block
body = ""
self._decide_mode(body, final=True)
+3 -1
View File
@@ -189,7 +189,9 @@ turn the other person talked over.
- The LLM receives the brief, the disclosure, the guardrails and the
conversation. It replies with `SAY:`, `ACTION: none | end_call | escalate` and
an optional `OUTCOME:` line, streamed so the first sentence is synthesized
while the rest is written.
while the rest is written. Anything ahead of that format — such as reasoning
from a chat template that prefills the opening tag into the prompt — is held
back and never spoken or recorded.
- Replies are synthesized with the streaming TTS pipeline in the chosen voice,
resampled to 8 kHz μ-law and sent as 20 ms frames. On barge-in (about 200 ms
of the other person's speech while the agent talks), VoiceStudio sends Twilio
+88
View File
@@ -301,6 +301,94 @@ def test_reply_parser_json_plain_and_reasoning_blocks():
assert p.action == "none"
def test_reply_parser_never_speaks_prefilled_thinking():
"""Fail-before/pass-after for #2428: a chat template that prefills the
opening tag into the prompt leaves only the closing tag on the wire.
The parser used to settle on plain at the first non-format characters
and spoke the reasoning — to the person on the phone, in the user's
voice — while the system prompt requires replies to start with SAY:."""
close = "<" + "/think>"
p = M_agent().ReplyParser()
out = ""
for d in ["The caller", " asked for the booking name. The brief says Palash. ",
"I should confirm and end.\n\n" + close +
"\n\nSAY: It's under Palash. Thank you, goodbye.\nACTION: end_call"]:
out += p.feed(d)
out += p.finish()
assert out.strip() == "It's under Palash. Thank you, goodbye."
assert "booking name" not in out # the reasoning never streamed
assert "brief says" not in out
assert "SAY:" not in out and "ACTION:" not in out
assert p.action == "end_call"
def test_reply_parser_drops_prefilled_thinking_that_never_closes():
"""A model that streams no closing tag either is still bounded by a
line-start SAY: — everything before it was thinking (#2428)."""
p = M_agent().ReplyParser()
out = ""
for d in ["I should confirm the booking. ", "Then end the call.\n",
"SAY: It's under Palash.\nACTION: none"]:
out += p.feed(d)
out += p.finish()
assert out.strip() == "It's under Palash."
assert "booking" not in out and "SAY:" not in out
assert p.action == "none"
def test_reply_parser_speaks_an_unformatted_reply_only_at_finish():
"""A reply with neither marker may be prefilled thinking, so it cannot
be spoken while it streams (#2428) — but it is still spoken: finish()
settles it as plain text, as it always did. Format-following replies
keep first-sentence streaming (see the two tests above)."""
p = M_agent().ReplyParser()
assert p.feed("Sorry, I cannot do that.") == ""
assert p.finish().strip() == "Sorry, I cannot do that."
assert p.action == "none"
def test_reply_parser_drops_prefilled_thinking_at_a_lowercase_boundary():
"""The line-start boundary is case-insensitive like SAY itself: a
lowercase `say:` must end prefilled thinking too, or finish() would
speak the reasoning as plain (#2428 review)."""
p = M_agent().ReplyParser()
out = ""
for d in ["I should confirm the booking.\n", "say: It's under Palash.\nACTION: none"]:
out += p.feed(d)
out += p.finish()
assert out.strip() == "It's under Palash."
assert "booking" not in out
assert p.action == "none"
def test_a_json_reply_is_not_resliced_by_a_trailing_say_line():
"""The line-start boundary applies only while no mode is chosen: a
body already streaming as json must not be re-sliced at finish() by a
stray SAY: line, or the say is dropped (#2431 review)."""
p = M_agent().ReplyParser()
body = '{"say": "Sure thing.", "action": "none"}\nSAY: stray line'
assert p.feed(body) == ""
assert p.finish() == "Sure thing."
assert p.action == "none"
def test_a_draft_say_line_inside_reasoning_is_not_spoken():
"""Unclosed reasoning may draft a `SAY:` line; only the closing tag —
or finish, if none ever comes — may establish the boundary, so a draft
must never reach the caller mid-stream (#2428 review)."""
close = "<" + "/think>"
p = M_agent().ReplyParser()
out = ""
for d in ["I'll reply now.\nSAY: draft, do not speak\n",
close + "\n\nSAY: It's under Palash.\nACTION: none"]:
out += p.feed(d)
assert "draft" not in out # never spoken, at any point in the stream
out += p.finish()
assert out.strip() == "It's under Palash."
assert "reply now" not in out and "draft" not in out
assert p.action == "none"
def test_guard_blocks_card_and_unknown_id_numbers_but_allows_brief_numbers():
agent = M_agent()
brief = "Callback number 415 555 0123 4."