chore: auto-commit from dashboard

This commit is contained in:
2026-08-22 08:40:56 +00:00
parent 4d2dbfb8a7
commit fe500a7227
35 changed files with 1372 additions and 144 deletions

View File

@@ -94,6 +94,31 @@ def expand_numbers_ro(text: str) -> str:
return _NUM_TOKEN.sub(_sub, text)
def _decimal_to_en(s: str) -> str:
"""Convert decimal string 'X.Y' to English words ('3.14' -> 'three point one four')."""
int_part, dec_part = s.split('.', 1)
int_words = num2words(int(int_part), lang='en')
dec_words = ' '.join(num2words(int(d), lang='en') for d in dec_part)
return f"{int_words} point {dec_words}"
def expand_numbers_en(text: str) -> str:
"""Expand bare numeric tokens to English words.
Mirrors expand_numbers_ro for the pocket-tts (English-only) path.
Must run after normalize_thousands so Romanian-style grouped
integers ("105.300") read as one magnitude ("one hundred five
thousand three hundred") instead of a decimal ("105.3").
"""
def _sub(match: re.Match) -> str:
token = match.group(1)
if '.' in token:
return _decimal_to_en(token)
return num2words(int(token), lang='en')
return _NUM_TOKEN.sub(_sub, text)
# ---------- Thousands separator ----------
# Romanian uses dot or space as thousands separator: 384.000 / 384 000. The
@@ -268,8 +293,8 @@ def expand_currency(text: str) -> str:
# ---------- Symbols ----------
_SYMBOL_WORDS = {
'ro': {'%': ' la sută', '&': ' și ', '@': ' la ', '°': ' grade'},
'en': {'%': ' percent', '&': ' and ', '@': ' at ', '°': ' degrees'},
'ro': {'%': ' la sută', '&': ' și ', '@': ' la ', '°': ' grade', '~': ' aproximativ '},
'en': {'%': ' percent', '&': ' and ', '@': ' at ', '°': ' degrees', '~': ' about '},
}
@@ -322,11 +347,14 @@ def expand_for_tts(text: str, lang: str = 'ro') -> str:
"Restul l-am scris în chat." suffix from normalize_for_tts() would be
misleading (no live chat mirror exists for that flow).
The RO-specific expansions (abbreviations, thousands, time, currency,
units, numbers-to-words) only make sense for Romanian text — for other
languages (e.g. English text routed to pocket-tts, which is
English-only) they'd inject Romanian words/diacritics into text the
target engine can't speak. Skip them when lang != 'ro'.
The RO-specific expansions (abbreviations, time, currency wording,
units) only make sense for Romanian text — for other languages (e.g.
English text routed to pocket-tts, which is English-only) they'd
inject Romanian words/diacritics into text the target engine can't
speak, so they're skipped when lang != 'ro'. Thousands-grouping and
numbers-to-words still run for lang == 'en', with English wording,
since replies on English turns often carry Romanian-formatted figures
("105.300 lei") that pocket-tts would otherwise misread as decimals.
"""
text = strip_markdown(text)
text = sanitize_punctuation(text)
@@ -337,6 +365,16 @@ def expand_for_tts(text: str, lang: str = 'ro') -> str:
text = expand_currency(text)
text = expand_units(text)
text = expand_numbers_ro(text)
elif lang == 'en':
# RO-specific expansions (abbreviations, time, currency wording,
# units) are skipped here — see docstring above. But bare numbers
# still need expanding: replies to Marius often carry Romanian
# thousands-grouped figures ("105.300 lei") even on English turns,
# and pocket-tts reads the dot as an English decimal point if left
# alone ("one hundred five point three" instead of "one hundred
# five thousand three hundred").
text = normalize_thousands(text)
text = expand_numbers_en(text)
text = expand_symbols(text, lang=lang)
return text.strip()