chore: auto-commit from dashboard
This commit is contained in:
@@ -94,6 +94,31 @@ def expand_numbers_ro(text: str) -> str:
|
||||
return _NUM_TOKEN.sub(_sub, text)
|
||||
|
||||
|
||||
def _decimal_to_en(s: str) -> str:
|
||||
"""Convert decimal string 'X.Y' to English words ('3.14' -> 'three point one four')."""
|
||||
int_part, dec_part = s.split('.', 1)
|
||||
int_words = num2words(int(int_part), lang='en')
|
||||
dec_words = ' '.join(num2words(int(d), lang='en') for d in dec_part)
|
||||
return f"{int_words} point {dec_words}"
|
||||
|
||||
|
||||
def expand_numbers_en(text: str) -> str:
|
||||
"""Expand bare numeric tokens to English words.
|
||||
|
||||
Mirrors expand_numbers_ro for the pocket-tts (English-only) path.
|
||||
Must run after normalize_thousands so Romanian-style grouped
|
||||
integers ("105.300") read as one magnitude ("one hundred five
|
||||
thousand three hundred") instead of a decimal ("105.3").
|
||||
"""
|
||||
def _sub(match: re.Match) -> str:
|
||||
token = match.group(1)
|
||||
if '.' in token:
|
||||
return _decimal_to_en(token)
|
||||
return num2words(int(token), lang='en')
|
||||
|
||||
return _NUM_TOKEN.sub(_sub, text)
|
||||
|
||||
|
||||
# ---------- Thousands separator ----------
|
||||
|
||||
# Romanian uses dot or space as thousands separator: 384.000 / 384 000. The
|
||||
@@ -268,8 +293,8 @@ def expand_currency(text: str) -> str:
|
||||
# ---------- Symbols ----------
|
||||
|
||||
_SYMBOL_WORDS = {
|
||||
'ro': {'%': ' la sută', '&': ' și ', '@': ' la ', '°': ' grade'},
|
||||
'en': {'%': ' percent', '&': ' and ', '@': ' at ', '°': ' degrees'},
|
||||
'ro': {'%': ' la sută', '&': ' și ', '@': ' la ', '°': ' grade', '~': ' aproximativ '},
|
||||
'en': {'%': ' percent', '&': ' and ', '@': ' at ', '°': ' degrees', '~': ' about '},
|
||||
}
|
||||
|
||||
|
||||
@@ -322,11 +347,14 @@ def expand_for_tts(text: str, lang: str = 'ro') -> str:
|
||||
"Restul l-am scris în chat." suffix from normalize_for_tts() would be
|
||||
misleading (no live chat mirror exists for that flow).
|
||||
|
||||
The RO-specific expansions (abbreviations, thousands, time, currency,
|
||||
units, numbers-to-words) only make sense for Romanian text — for other
|
||||
languages (e.g. English text routed to pocket-tts, which is
|
||||
English-only) they'd inject Romanian words/diacritics into text the
|
||||
target engine can't speak. Skip them when lang != 'ro'.
|
||||
The RO-specific expansions (abbreviations, time, currency wording,
|
||||
units) only make sense for Romanian text — for other languages (e.g.
|
||||
English text routed to pocket-tts, which is English-only) they'd
|
||||
inject Romanian words/diacritics into text the target engine can't
|
||||
speak, so they're skipped when lang != 'ro'. Thousands-grouping and
|
||||
numbers-to-words still run for lang == 'en', with English wording,
|
||||
since replies on English turns often carry Romanian-formatted figures
|
||||
("105.300 lei") that pocket-tts would otherwise misread as decimals.
|
||||
"""
|
||||
text = strip_markdown(text)
|
||||
text = sanitize_punctuation(text)
|
||||
@@ -337,6 +365,16 @@ def expand_for_tts(text: str, lang: str = 'ro') -> str:
|
||||
text = expand_currency(text)
|
||||
text = expand_units(text)
|
||||
text = expand_numbers_ro(text)
|
||||
elif lang == 'en':
|
||||
# RO-specific expansions (abbreviations, time, currency wording,
|
||||
# units) are skipped here — see docstring above. But bare numbers
|
||||
# still need expanding: replies to Marius often carry Romanian
|
||||
# thousands-grouped figures ("105.300 lei") even on English turns,
|
||||
# and pocket-tts reads the dot as an English decimal point if left
|
||||
# alone ("one hundred five point three" instead of "one hundred
|
||||
# five thousand three hundred").
|
||||
text = normalize_thousands(text)
|
||||
text = expand_numbers_en(text)
|
||||
text = expand_symbols(text, lang=lang)
|
||||
return text.strip()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user