mirror of
https://github.com/debpalash/VoiceStudio.git
synced 2026-10-02 09:34:38 +08:00
Merge commit 'refs/pull/2431/head' of github.com:debpalash/VoiceStudio into triage/clone-asr-llm
This commit is contained in:
@@ -98,6 +98,7 @@ metadata and the backend fallback mirror it.
|
||||
- MLX Qwen3-TTS receives the selected language correctly; MeloTTS explains missing text resources without downloading during generation (#2419)
|
||||
- The language picker offers only the languages each MLX-Audio model supports (Kokoro, CSM, Qwen3-TTS, Dia, Chatterbox, MeloTTS, OuteTTS), per their model cards, instead of every language (#977)
|
||||
|
||||
- The call agent holds back reasoning from chat templates that prefill the opening tag instead of speaking it to the caller (#2428) — thanks @swadhinbiswas!
|
||||
- The Enter that confirms Korean, Japanese or Chinese input no longer also submits project renames, language search, pronunciation, worker or MCP fields (#2338) — thanks @HEOJUNFO!
|
||||
- English text normalization speaks a dollar amount followed by a period or comma ("It costs $5.") instead of leaving the digits (#2390) — thanks @kevin9327!
|
||||
- Voice Design sends descriptions unchanged to free-text engines and restores the original design when reopening a take (#2401) — thanks @CauaMatheus and @dominikj-cf!
|
||||
|
||||
@@ -190,6 +190,16 @@ def guard_sensitive(sentence: str, brief: str) -> str:
|
||||
# ── Structured LLM reply ────────────────────────────────────────────────────
|
||||
|
||||
_THINK_OPEN_RE = re.compile(r"^\s*<(think|thinking|reasoning)>", re.IGNORECASE)
|
||||
# A chat template that prefills the opening tag into the prompt leaves only
|
||||
# the closing one on the wire (#2428) — there is no opening tag to match.
|
||||
_THINK_CLOSE_RE = re.compile(r"</(?:think|thinking|reasoning)>", re.IGNORECASE)
|
||||
# A line-start SAY: later in an undecided body marks where prefilled
|
||||
# thinking ends, for a model that streams no closing tag either (#2428).
|
||||
# Case-insensitive like _SAY_RE: a lowercase boundary must not leave the
|
||||
# reasoning to finish()'s plain path. Only consulted at finish — a draft
|
||||
# "SAY:" line *inside* reasoning must not switch to tagged while a
|
||||
# closing tag may still arrive.
|
||||
_SAY_LINE_RE = re.compile(r"(?m)^[ \t]*SAY\s*:", re.IGNORECASE)
|
||||
_TAG_RE = re.compile(r"(?:^|\s)(ACTION|OUTCOME)\s*:", re.IGNORECASE)
|
||||
_SAY_RE = re.compile(r"^\s*SAY\s*:\s*", re.IGNORECASE)
|
||||
_ACTION_VALUE_RE = re.compile(r"ACTION\s*:\s*([A-Za-z_\- ]+)", re.IGNORECASE)
|
||||
@@ -226,7 +236,7 @@ class ReplyParser:
|
||||
self.action = "none"
|
||||
self.outcome: str | None = None
|
||||
|
||||
def _body(self) -> str | None:
|
||||
def _body(self, final: bool = False) -> str | None:
|
||||
text = self.raw
|
||||
match = _THINK_OPEN_RE.match(text)
|
||||
if match:
|
||||
@@ -234,6 +244,25 @@ class ReplyParser:
|
||||
if not close:
|
||||
return None
|
||||
text = text[close.end():]
|
||||
else:
|
||||
# No opening tag: either there is no reasoning at all, or the
|
||||
# chat template prefilled the opening tag into the prompt and
|
||||
# the model streams only …SAY: (#2428).
|
||||
close = _THINK_CLOSE_RE.search(text)
|
||||
if close:
|
||||
text = text[close.end():]
|
||||
# A line-start SAY: marks where thinking ends for the model that
|
||||
# never sent a closing tag either (#2428) — but only at finish, and
|
||||
# only while no mode has been chosen (#2431 review): a draft "SAY:"
|
||||
# line inside still-streaming reasoning must not flip the parser to
|
||||
# tagged before a closing tag establishes the real boundary, and a
|
||||
# body already streaming as tagged/json must never be re-sliced
|
||||
# here, or finish() would discard text ``_emitted`` already counts
|
||||
# (or, for json, drop a say that parses). While streaming everything
|
||||
# without a format marker is held anyway, so waiting costs nothing.
|
||||
say = _SAY_LINE_RE.search(text) if final and self.mode in (None, "plain") else None
|
||||
if say:
|
||||
text = text[say.start():]
|
||||
return text.lstrip()
|
||||
|
||||
def _decide_mode(self, body: str, final: bool) -> None:
|
||||
@@ -243,8 +272,15 @@ class ReplyParser:
|
||||
self.mode = "json"
|
||||
elif _SAY_RE.match(body):
|
||||
self.mode = "tagged"
|
||||
elif final or len(body) >= 4 or not "say:".startswith(body[:4].lower()):
|
||||
elif final:
|
||||
self.mode = "plain"
|
||||
# Not final and neither format marker: hold. The system prompt
|
||||
# requires replies to start with SAY: or {, so anything else may be
|
||||
# prefilled thinking (#2428) — the first spoken sentence cannot be
|
||||
# taken back, and this text reaches a third party on the phone.
|
||||
# ``_body`` reveals the format when the closing tag or a later
|
||||
# line-start SAY: arrives; ``finish`` settles a reply with neither
|
||||
# as plain, as it always did.
|
||||
|
||||
def _say(self, body: str, final: bool) -> str:
|
||||
if self.mode == "json":
|
||||
@@ -269,7 +305,7 @@ class ReplyParser:
|
||||
|
||||
def feed(self, delta: str) -> str:
|
||||
self.raw += delta or ""
|
||||
body = self._body()
|
||||
body = self._body(final=False)
|
||||
if body is None:
|
||||
return ""
|
||||
self._decide_mode(body, final=False)
|
||||
@@ -278,7 +314,7 @@ class ReplyParser:
|
||||
return self._take(self._say(body, final=False))
|
||||
|
||||
def finish(self) -> str:
|
||||
body = self._body()
|
||||
body = self._body(final=True)
|
||||
if body is None: # never left the reasoning block
|
||||
body = ""
|
||||
self._decide_mode(body, final=True)
|
||||
|
||||
@@ -189,7 +189,9 @@ turn the other person talked over.
|
||||
- The LLM receives the brief, the disclosure, the guardrails and the
|
||||
conversation. It replies with `SAY:`, `ACTION: none | end_call | escalate` and
|
||||
an optional `OUTCOME:` line, streamed so the first sentence is synthesized
|
||||
while the rest is written.
|
||||
while the rest is written. Anything ahead of that format — such as reasoning
|
||||
from a chat template that prefills the opening tag into the prompt — is held
|
||||
back and never spoken or recorded.
|
||||
- Replies are synthesized with the streaming TTS pipeline in the chosen voice,
|
||||
resampled to 8 kHz μ-law and sent as 20 ms frames. On barge-in (about 200 ms
|
||||
of the other person's speech while the agent talks), VoiceStudio sends Twilio
|
||||
|
||||
@@ -301,6 +301,94 @@ def test_reply_parser_json_plain_and_reasoning_blocks():
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_reply_parser_never_speaks_prefilled_thinking():
|
||||
"""Fail-before/pass-after for #2428: a chat template that prefills the
|
||||
opening tag into the prompt leaves only the closing tag on the wire.
|
||||
The parser used to settle on plain at the first non-format characters
|
||||
and spoke the reasoning — to the person on the phone, in the user's
|
||||
voice — while the system prompt requires replies to start with SAY:."""
|
||||
close = "<" + "/think>"
|
||||
p = M_agent().ReplyParser()
|
||||
out = ""
|
||||
for d in ["The caller", " asked for the booking name. The brief says Palash. ",
|
||||
"I should confirm and end.\n\n" + close +
|
||||
"\n\nSAY: It's under Palash. Thank you, goodbye.\nACTION: end_call"]:
|
||||
out += p.feed(d)
|
||||
out += p.finish()
|
||||
assert out.strip() == "It's under Palash. Thank you, goodbye."
|
||||
assert "booking name" not in out # the reasoning never streamed
|
||||
assert "brief says" not in out
|
||||
assert "SAY:" not in out and "ACTION:" not in out
|
||||
assert p.action == "end_call"
|
||||
|
||||
|
||||
def test_reply_parser_drops_prefilled_thinking_that_never_closes():
|
||||
"""A model that streams no closing tag either is still bounded by a
|
||||
line-start SAY: — everything before it was thinking (#2428)."""
|
||||
p = M_agent().ReplyParser()
|
||||
out = ""
|
||||
for d in ["I should confirm the booking. ", "Then end the call.\n",
|
||||
"SAY: It's under Palash.\nACTION: none"]:
|
||||
out += p.feed(d)
|
||||
out += p.finish()
|
||||
assert out.strip() == "It's under Palash."
|
||||
assert "booking" not in out and "SAY:" not in out
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_reply_parser_speaks_an_unformatted_reply_only_at_finish():
|
||||
"""A reply with neither marker may be prefilled thinking, so it cannot
|
||||
be spoken while it streams (#2428) — but it is still spoken: finish()
|
||||
settles it as plain text, as it always did. Format-following replies
|
||||
keep first-sentence streaming (see the two tests above)."""
|
||||
p = M_agent().ReplyParser()
|
||||
assert p.feed("Sorry, I cannot do that.") == ""
|
||||
assert p.finish().strip() == "Sorry, I cannot do that."
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_reply_parser_drops_prefilled_thinking_at_a_lowercase_boundary():
|
||||
"""The line-start boundary is case-insensitive like SAY itself: a
|
||||
lowercase `say:` must end prefilled thinking too, or finish() would
|
||||
speak the reasoning as plain (#2428 review)."""
|
||||
p = M_agent().ReplyParser()
|
||||
out = ""
|
||||
for d in ["I should confirm the booking.\n", "say: It's under Palash.\nACTION: none"]:
|
||||
out += p.feed(d)
|
||||
out += p.finish()
|
||||
assert out.strip() == "It's under Palash."
|
||||
assert "booking" not in out
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_a_json_reply_is_not_resliced_by_a_trailing_say_line():
|
||||
"""The line-start boundary applies only while no mode is chosen: a
|
||||
body already streaming as json must not be re-sliced at finish() by a
|
||||
stray SAY: line, or the say is dropped (#2431 review)."""
|
||||
p = M_agent().ReplyParser()
|
||||
body = '{"say": "Sure thing.", "action": "none"}\nSAY: stray line'
|
||||
assert p.feed(body) == ""
|
||||
assert p.finish() == "Sure thing."
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_a_draft_say_line_inside_reasoning_is_not_spoken():
|
||||
"""Unclosed reasoning may draft a `SAY:` line; only the closing tag —
|
||||
or finish, if none ever comes — may establish the boundary, so a draft
|
||||
must never reach the caller mid-stream (#2428 review)."""
|
||||
close = "<" + "/think>"
|
||||
p = M_agent().ReplyParser()
|
||||
out = ""
|
||||
for d in ["I'll reply now.\nSAY: draft, do not speak\n",
|
||||
close + "\n\nSAY: It's under Palash.\nACTION: none"]:
|
||||
out += p.feed(d)
|
||||
assert "draft" not in out # never spoken, at any point in the stream
|
||||
out += p.finish()
|
||||
assert out.strip() == "It's under Palash."
|
||||
assert "reply now" not in out and "draft" not in out
|
||||
assert p.action == "none"
|
||||
|
||||
|
||||
def test_guard_blocks_card_and_unknown_id_numbers_but_allows_brief_numbers():
|
||||
agent = M_agent()
|
||||
brief = "Callback number 415 555 0123 4."
|
||||
|
||||
Reference in New Issue
Block a user