Merge PR #2455: bind Unicode pronunciation matches to their lexicon entry (#2450)

Co-Authored-By: Claude Sonnet 5.5 <noreply@anthropic.com>
This commit is contained in:
Palash Debnath
2026-10-01 18:43:19 +05:30
co-authored by Claude Sonnet 5.5
3 changed files with 48 additions and 7 deletions
+11 -7
View File
@@ -84,7 +84,7 @@ def _boundary_suffix(key: str) -> str:
def _compile(lexicon: dict[str, str]) -> tuple[Optional[re.Pattern], dict[str, str]]:
"""Build the single alternation regex + a casefold→respelling lookup.
"""Build the single alternation regex + matched-group→respelling lookup.
Keys are sorted longest-first so an overlapping longer key (``Dr. Smith``)
is tried before a shorter one (``Dr``). Each alternative carries its own
@@ -92,13 +92,17 @@ def _compile(lexicon: dict[str, str]) -> tuple[Optional[re.Pattern], dict[str, s
punctuation-edged key (``Dr.``) matchable while still protecting a
letter-edged key (``cat``) from partial hits inside ``category``.
"""
keys = sorted(lexicon.keys(), key=len, reverse=True)
# Equal-length case variants retain the existing last-entry precedence.
keys = sorted(reversed(lexicon), key=len, reverse=True)
if not keys:
return None, {}
# casefold (not lower) for robust Unicode case-insensitive lookup.
lookup = {k.casefold(): lexicon[k] for k in keys}
alts = [f"{_boundary_prefix(k)}{re.escape(k)}{_boundary_suffix(k)}" for k in keys]
# No capturing groups, no nested quantifiers — pure literal alternation.
# IGNORECASE and casefold have different Unicode equivalence classes:
# e.g. 'i' matches dotless 'ı', while Straße/STRASSE do not regex-match.
# Bind each literal to its replacement rather than folding matched text.
lookup = {f"term_{i}": lexicon[k] for i, k in enumerate(keys)}
alts = [f"(?P<term_{i}>{_boundary_prefix(k)}{re.escape(k)}{_boundary_suffix(k)})"
for i, k in enumerate(keys)]
# No nested quantifiers — pure literal alternation.
pattern = re.compile("(?:" + "|".join(alts) + ")", re.IGNORECASE)
return pattern, lookup
@@ -122,7 +126,7 @@ def apply_lexicon(text: str, lexicon: Optional[dict]) -> str:
return text
def _repl(m: re.Match) -> str:
return lookup.get(m.group(0).casefold(), m.group(0))
return lookup[m.lastgroup]
return pattern.sub(_repl, text)
+2
View File
@@ -62,6 +62,8 @@ for your engine below.
dictionary. Not expression, but often what a "it says this weirdly" problem
actually needs.
Pronunciation dictionary matching uses Unicode case-insensitive literal matches. Each matched term uses its own respelling; distinct terms such as Straße and STRASSE can have different respellings. Longer terms win overlaps, and later equal-length case variants retain precedence.
### Default engine (VoiceStudio)
**Non-verbal tags.** The bundled model natively tokenizes 13 reaction tags
@@ -0,0 +1,35 @@
"""Every literal matched by the lexicon must use that entry's replacement."""
import pytest
@pytest.mark.parametrize('term,text', [('GIF', 'gıf'), ('index', 'İNDEX'), ('seek', 'ſeek'), ('kelvin', 'KELVIN')])
def test_regex_case_matches_use_the_matched_entry(term, text):
from services.pronunciation import apply_lexicon
assert apply_lexicon('Say ' + text + '.', {term: 'replacement'}) == 'Say replacement.'
def test_distinct_literals_do_not_share_a_casefold_replacement():
from services.pronunciation import apply_lexicon
assert apply_lexicon('Straße STRASSE', {'Straße': 'street', 'STRASSE': 'avenue'}) == 'street avenue'
assert apply_lexicon('Dr. Smith Dr.', {'Dr.': 'Doctor', 'Dr. Smith': 'Professor Smith'}) == 'Professor Smith Doctor'
def test_case_variants_keep_last_entry_precedence():
from services.pronunciation import apply_lexicon
assert apply_lexicon('GIF gif', {'GIF': 'global', 'gif': 'local'}) == 'local local'
assert apply_lexicon('GIF gif', {'gif': 'local', 'GIF': 'global'}) == 'global global'
def test_language_specific_case_variant_overrides_global():
from services.pronunciation import apply_lexicon
from services.pronunciation import entries_for_language
entries = [
{'term': 'GIF', 'replacement': 'global', 'type': 'respelling', 'language': '*', 'enabled': 1},
{'term': 'gif', 'replacement': 'local', 'type': 'respelling', 'language': 'en', 'enabled': 1},
]
assert apply_lexicon('GIF gif', entries_for_language(entries, 'en-US')) == 'local local'