mirror of
https://github.com/debpalash/VoiceStudio.git
synced 2026-10-02 09:34:38 +08:00
Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
@@ -84,7 +84,7 @@ def _boundary_suffix(key: str) -> str:
|
||||
|
||||
|
||||
def _compile(lexicon: dict[str, str]) -> tuple[Optional[re.Pattern], dict[str, str]]:
|
||||
"""Build the single alternation regex + a casefold→respelling lookup.
|
||||
"""Build the single alternation regex + matched-group→respelling lookup.
|
||||
|
||||
Keys are sorted longest-first so an overlapping longer key (``Dr. Smith``)
|
||||
is tried before a shorter one (``Dr``). Each alternative carries its own
|
||||
@@ -92,13 +92,17 @@ def _compile(lexicon: dict[str, str]) -> tuple[Optional[re.Pattern], dict[str, s
|
||||
punctuation-edged key (``Dr.``) matchable while still protecting a
|
||||
letter-edged key (``cat``) from partial hits inside ``category``.
|
||||
"""
|
||||
keys = sorted(lexicon.keys(), key=len, reverse=True)
|
||||
# Equal-length case variants retain the existing last-entry precedence.
|
||||
keys = sorted(reversed(lexicon), key=len, reverse=True)
|
||||
if not keys:
|
||||
return None, {}
|
||||
# casefold (not lower) for robust Unicode case-insensitive lookup.
|
||||
lookup = {k.casefold(): lexicon[k] for k in keys}
|
||||
alts = [f"{_boundary_prefix(k)}{re.escape(k)}{_boundary_suffix(k)}" for k in keys]
|
||||
# No capturing groups, no nested quantifiers — pure literal alternation.
|
||||
# IGNORECASE and casefold have different Unicode equivalence classes:
|
||||
# e.g. 'i' matches dotless 'ı', while Straße/STRASSE do not regex-match.
|
||||
# Bind each literal to its replacement rather than folding matched text.
|
||||
lookup = {f"term_{i}": lexicon[k] for i, k in enumerate(keys)}
|
||||
alts = [f"(?P<term_{i}>{_boundary_prefix(k)}{re.escape(k)}{_boundary_suffix(k)})"
|
||||
for i, k in enumerate(keys)]
|
||||
# No nested quantifiers — pure literal alternation.
|
||||
pattern = re.compile("(?:" + "|".join(alts) + ")", re.IGNORECASE)
|
||||
return pattern, lookup
|
||||
|
||||
@@ -122,7 +126,7 @@ def apply_lexicon(text: str, lexicon: Optional[dict]) -> str:
|
||||
return text
|
||||
|
||||
def _repl(m: re.Match) -> str:
|
||||
return lookup.get(m.group(0).casefold(), m.group(0))
|
||||
return lookup[m.lastgroup]
|
||||
|
||||
return pattern.sub(_repl, text)
|
||||
|
||||
|
||||
@@ -62,6 +62,8 @@ for your engine below.
|
||||
dictionary. Not expression, but often what a "it says this weirdly" problem
|
||||
actually needs.
|
||||
|
||||
Pronunciation dictionary matching uses Unicode case-insensitive literal matches. Each matched term uses its own respelling; distinct terms such as Straße and STRASSE can have different respellings. Longer terms win overlaps, and later equal-length case variants retain precedence.
|
||||
|
||||
### Default engine (VoiceStudio)
|
||||
|
||||
**Non-verbal tags.** The bundled model natively tokenizes 13 reaction tags
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
"""Every literal matched by the lexicon must use that entry's replacement."""
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.mark.parametrize('term,text', [('GIF', 'gıf'), ('index', 'İNDEX'), ('seek', 'ſeek'), ('kelvin', 'KELVIN')])
|
||||
def test_regex_case_matches_use_the_matched_entry(term, text):
|
||||
from services.pronunciation import apply_lexicon
|
||||
|
||||
assert apply_lexicon('Say ' + text + '.', {term: 'replacement'}) == 'Say replacement.'
|
||||
|
||||
|
||||
def test_distinct_literals_do_not_share_a_casefold_replacement():
|
||||
from services.pronunciation import apply_lexicon
|
||||
|
||||
assert apply_lexicon('Straße STRASSE', {'Straße': 'street', 'STRASSE': 'avenue'}) == 'street avenue'
|
||||
assert apply_lexicon('Dr. Smith Dr.', {'Dr.': 'Doctor', 'Dr. Smith': 'Professor Smith'}) == 'Professor Smith Doctor'
|
||||
|
||||
|
||||
def test_case_variants_keep_last_entry_precedence():
|
||||
from services.pronunciation import apply_lexicon
|
||||
|
||||
assert apply_lexicon('GIF gif', {'GIF': 'global', 'gif': 'local'}) == 'local local'
|
||||
assert apply_lexicon('GIF gif', {'gif': 'local', 'GIF': 'global'}) == 'global global'
|
||||
|
||||
|
||||
def test_language_specific_case_variant_overrides_global():
|
||||
from services.pronunciation import apply_lexicon
|
||||
|
||||
from services.pronunciation import entries_for_language
|
||||
|
||||
entries = [
|
||||
{'term': 'GIF', 'replacement': 'global', 'type': 'respelling', 'language': '*', 'enabled': 1},
|
||||
{'term': 'gif', 'replacement': 'local', 'type': 'respelling', 'language': 'en', 'enabled': 1},
|
||||
]
|
||||
assert apply_lexicon('GIF gif', entries_for_language(entries, 'en-US')) == 'local local'
|
||||
Reference in New Issue
Block a user