fix(voice): normalizare TTS lang-aware + cap 400 chars mutat per-request Supertonic
Două cauze pentru audio mut pe pocket-tts după fix-ul [tts-lang:en]:
1. normalize_for_tts expanda numerele/orele în cuvinte românești chiar și
în text englezesc ("It's 9:46 PM" → "nouă și patruzeci și șase de
minute") → diacritice → pocket-tts respingea clauzele. push_text rezolvă
acum lang din engine-ul vocii curente (engine_for_voice) și sare
expansiunile RO când lang != 'ro'. [scris de Echo în sesiunea Discord]
2. sanitize_for_supertonic (cap 400 chars — limită ONNX per request) rula
pe textul întreg în normalize, înainte de segmentarea în clauze —
răspunsurile lungi erau retezate mid-word și trunchierea la 200 cuvinte
cu sufixul "Restul l-am scris în chat" nu se mai aplica. Separat
map_tts_punctuation (fără cap, folosit de normalize) de
sanitize_for_supertonic (cap, aplicat în _synthesize_supertonic).
Include vocea clonată "Marius 4" (tts_voices.json + default în config).
Fixează test_truncate_exactly_200_words_unchanged și
test_truncate_over_200_words_appends_suffix.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
This commit is contained in:
@@ -104,7 +104,7 @@
|
||||
"949388626146517022"
|
||||
],
|
||||
"user_name": "Marius",
|
||||
"default_voice": "Marius 3",
|
||||
"default_voice": "Marius 4",
|
||||
"auto_leave_minutes": 5,
|
||||
"stt_model": "/home/moltbot/echo-core/models/whisper-small-ro-cv11-int8"
|
||||
},
|
||||
|
||||
@@ -267,17 +267,25 @@ def expand_currency(text: str) -> str:
|
||||
|
||||
# ---------- Symbols ----------
|
||||
|
||||
def expand_symbols(text: str) -> str:
|
||||
"""Replace common symbols with their Romanian spoken form."""
|
||||
text = text.replace('%', ' la sută')
|
||||
text = text.replace('&', ' și ')
|
||||
text = text.replace('@', ' la ')
|
||||
text = text.replace('°', ' grade')
|
||||
_SYMBOL_WORDS = {
|
||||
'ro': {'%': ' la sută', '&': ' și ', '@': ' la ', '°': ' grade'},
|
||||
'en': {'%': ' percent', '&': ' and ', '@': ' at ', '°': ' degrees'},
|
||||
}
|
||||
|
||||
|
||||
def expand_symbols(text: str, lang: str = 'ro') -> str:
|
||||
"""Replace common symbols with their spoken form for the given language."""
|
||||
words = _SYMBOL_WORDS.get(lang, _SYMBOL_WORDS['ro'])
|
||||
for symbol, word in words.items():
|
||||
text = text.replace(symbol, word)
|
||||
text = re.sub(r'\s+', ' ', text).strip()
|
||||
return text
|
||||
|
||||
|
||||
from tools.tts import sanitize_for_supertonic as sanitize_punctuation
|
||||
# Punctuation-only mapping (fără cap de lungime) — cap-ul de 400 chars e
|
||||
# limita per-request Supertonic și se aplică în _synthesize_supertonic,
|
||||
# după segmentarea în clauze, nu aici pe textul întreg.
|
||||
from tools.tts import map_tts_punctuation as sanitize_punctuation
|
||||
|
||||
|
||||
# ---------- Abbreviations ----------
|
||||
@@ -306,34 +314,41 @@ _MAX_WORDS = 200
|
||||
_TRUNCATE_SUFFIX = "Restul l-am scris în chat."
|
||||
|
||||
|
||||
def expand_for_tts(text: str) -> str:
|
||||
def expand_for_tts(text: str, lang: str = 'ro') -> str:
|
||||
"""Apply the full normalization pipeline (markdown strip, abbreviations,
|
||||
numbers, currency, units, symbols) WITHOUT the 200-word truncation.
|
||||
|
||||
Use this for one-shot TTS generation (e.g. /audio command) where the
|
||||
"Restul l-am scris în chat." suffix from normalize_for_tts() would be
|
||||
misleading (no live chat mirror exists for that flow).
|
||||
|
||||
The RO-specific expansions (abbreviations, thousands, time, currency,
|
||||
units, numbers-to-words) only make sense for Romanian text — for other
|
||||
languages (e.g. English text routed to pocket-tts, which is
|
||||
English-only) they'd inject Romanian words/diacritics into text the
|
||||
target engine can't speak. Skip them when lang != 'ro'.
|
||||
"""
|
||||
text = strip_markdown(text)
|
||||
text = sanitize_punctuation(text)
|
||||
if lang == 'ro':
|
||||
text = expand_abbreviations(text)
|
||||
text = normalize_thousands(text)
|
||||
text = expand_time(text)
|
||||
text = expand_currency(text)
|
||||
text = expand_units(text)
|
||||
text = expand_numbers_ro(text)
|
||||
text = expand_symbols(text)
|
||||
text = expand_symbols(text, lang=lang)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def normalize_for_tts(text: str) -> str:
|
||||
def normalize_for_tts(text: str, lang: str = 'ro') -> str:
|
||||
"""Apply the full normalization pipeline and truncate to 200 words.
|
||||
|
||||
If the text exceeds 200 words, the first 200 are kept and the suffix
|
||||
"Restul l-am scris în chat." is appended so the listener knows the
|
||||
response continues in the text channel mirror.
|
||||
"""
|
||||
text = expand_for_tts(text)
|
||||
text = expand_for_tts(text, lang=lang)
|
||||
words = text.split()
|
||||
if len(words) > _MAX_WORDS:
|
||||
text = ' '.join(words[:_MAX_WORDS]) + f" {_TRUNCATE_SUFFIX}"
|
||||
|
||||
@@ -22,7 +22,7 @@ from typing import Iterator, List, Optional
|
||||
import discord
|
||||
|
||||
from src.voice.normalize import normalize_for_tts
|
||||
from tools.tts import synthesize
|
||||
from tools.tts import engine_for_voice, synthesize
|
||||
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
@@ -202,7 +202,14 @@ class TTSQueue:
|
||||
"""Normalize, segment into clauses, enqueue each clause for synthesis."""
|
||||
if not text:
|
||||
return
|
||||
cleaned = normalize_for_tts(text)
|
||||
# Resolve lang from the *current* voice_id (not self.lang, which is
|
||||
# set once at construction and goes stale on in-band voice swap —
|
||||
# see session.ttsq.voice_id mutation in discord_voice.py / pipeline.py).
|
||||
# pocket-tts is English-only, so RO number/time/currency expansion
|
||||
# must be skipped or the normalizer injects Romanian diacritics that
|
||||
# pocket-tts refuses to speak (silent dropped clause).
|
||||
lang = "en" if engine_for_voice(self.voice_id) == "pockettts" else "ro"
|
||||
cleaned = normalize_for_tts(text, lang=lang)
|
||||
n = 0
|
||||
for clause in clause_segments(cleaned):
|
||||
clause = clause.strip()
|
||||
|
||||
17
tools/tts.py
17
tools/tts.py
@@ -62,8 +62,14 @@ def _looks_romanian(text: str) -> bool:
|
||||
return any(ch in _RO_DIACRITICS for ch in text)
|
||||
|
||||
|
||||
def sanitize_for_supertonic(text: str) -> str:
|
||||
"""Replace Unicode punctuation and strip chars that crash Supertonic's ONNX model."""
|
||||
def map_tts_punctuation(text: str) -> str:
|
||||
"""Replace Unicode punctuation with ASCII and strip emoji — NO length cap.
|
||||
|
||||
Folosit și de pipeline-ul de normalizare voice (src/voice/normalize.py),
|
||||
care rulează pe textul întreg ÎNAINTE de segmentarea în clauze — un cap de
|
||||
lungime aici ar reteza răspunsurile lungi în mijlocul cuvântului și ar
|
||||
face moartă trunchierea la 200 de cuvinte din normalize_for_tts.
|
||||
"""
|
||||
for src, dst in _TTS_PUNCT_MAP.items():
|
||||
text = text.replace(src, dst)
|
||||
# Strip emoji and high-codepoint chars (keep ASCII printable + Latin/Romanian diacritice)
|
||||
@@ -74,7 +80,12 @@ def sanitize_for_supertonic(text: str) -> str:
|
||||
cleaned.append(ch)
|
||||
else:
|
||||
cleaned.append(' ')
|
||||
text = ' '.join(''.join(cleaned).split())
|
||||
return ' '.join(''.join(cleaned).split())
|
||||
|
||||
|
||||
def sanitize_for_supertonic(text: str) -> str:
|
||||
"""map_tts_punctuation + hard cap la limita ONNX Supertonic (per request)."""
|
||||
text = map_tts_punctuation(text)
|
||||
if len(text) > _MAX_TTS_CHARS:
|
||||
text = text[:_MAX_TTS_CHARS]
|
||||
return text
|
||||
|
||||
@@ -62,5 +62,10 @@
|
||||
"alba": {
|
||||
"engine": "pockettts",
|
||||
"voice_url": "alba"
|
||||
},
|
||||
"Marius 4": {
|
||||
"engine": "pockettts",
|
||||
"state_path": "models/voices/marius-4.safetensors",
|
||||
"owner": "Marius"
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user