dialogs-ru-tts / app.py
frappuccino's picture
Integrate rutextnorm: normalize toggle + numeric example
ec2687c verified
Raw
History Blame Contribute Delete
7.65 kB
"""Gradio demo for the Dialogs-RU expressive Russian TTS model (VITS2).
Free CPU demo. Type Russian text, pick a voice and an emotion, and the model
speaks. Stress (´) is placed automatically; you can also mark it yourself with
a ``+`` before the stressed vowel (e.g. "прив+ет").
"""
import os
# Keep native thread pools small and avoid duplicate-OpenMP aborts between
# torch and onnxruntime (ruaccent) on the small CPU Space.
os.environ.setdefault("OMP_NUM_THREADS", "2")
os.environ.setdefault("KMP_DUPLICATE_LIB_OK", "TRUE")
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
import gradio as gr
from tts import DialogsTTS, SPEAKERS, EMOTIONS
# ----------------------------------------------------------------------------
# Load the model once at startup.
# ----------------------------------------------------------------------------
TTS = DialogsTTS()
EMOJI = {
"neutral": "😐", "happy": "😊", "surprise": "😲", "arrogance": "😏",
"yawn": "🥱", "fear": "😨", "laughing": "😄", "whisper": "🤫",
"disgust": "🤢", "angry": "😠", "sad": "😢",
"tongue-twister": "👅", "poem": "📜",
}
SPEAKER_LABEL_TO_ID = {
f"{'👩' if g == 'female' else '👨'} {en} / {ru}": sid
for sid, (en, ru, g) in SPEAKERS.items()
}
EMOTION_LABEL_TO_ID = {
f"{EMOJI.get(en, '🎭')} {en} / {ru}": eid
for eid, (en, ru) in EMOTIONS.items()
}
SPEAKER_CHOICES = list(SPEAKER_LABEL_TO_ID.keys())
EMOTION_CHOICES = list(EMOTION_LABEL_TO_ID.keys())
DESCRIPTION = """
# 🎭 Dialogs-RU · Expressive Russian Text-to-Speech
A **VITS2** speech synthesizer trained on **[Dialogs](https://huggingface.co/datasets/langswap/dialogs-ru-emotional-conversations)**,
a studio-quality, expressive, conversational **Russian** speech corpus.
Choose one of **three studio voices** and one of **13 emotional styles**, type some Russian, and press **Generate**.
Runs on **CPU** — free for everyone to try. 🎉
- 📚 **Dataset:** [langswap/dialogs-ru-emotional-conversations](https://huggingface.co/datasets/langswap/dialogs-ru-emotional-conversations) · OpenRAIL
- 🧠 **Model weights:** [frappuccino/dialogs-ru-vits2](https://huggingface.co/frappuccino/dialogs-ru-vits2)
- 💻 **Training code:** [github.com/shigabeev/vits2-emotional](https://github.com/shigabeev/vits2-emotional)
- 📄 **Paper:** *Dialogs: a studio-quality expressive conversational Russian speech corpus for dialog assistants* (Shigabeev & Latyshev, Langswap)
"""
TIPS = """
### Tips
- **Numbers, dates, money, units & abbreviations are read out in full** —
normalized automatically with [`rutextnorm`](https://github.com/shigabeev/russian_tts_normalization)
(e.g. `7,5%` → «семь целых и пять десятых процента», `№5` → «номер пять»).
- **Stress matters in Russian.** Stress is placed automatically with
[`ruaccent`](https://github.com/Den4ikAI/ruaccent). To override it, put a
`+` right before the stressed vowel: `за́мок` → `з+амок`, `замо́к` → `зам+ок`.
- **Voices:** Masha & Sveta are female, Dima is male.
- **Styles** like *laughing*, *whisper*, *poem* or *tongue-twister* are rarer in
the data, so they are more subtle. *Neutral*, *happy* and *sad* are the strongest.
- This is a research proof-of-concept (UTMOS ≈ 3.36); expect lively, conversational
prosody rather than perfectly clean studio audio.
"""
EXAMPLES = [
["Привет! Меня зовут Маша. Чем могу помочь?", "👩 Masha / Маша", "😊 happy / радостный", True, True],
["Я так устала сегодня... совсем нет сил.", "👩 Sveta / Света", "😢 sad / грусть", True, True],
["В 2024 году цена выросла на 7,5%, до 1 500 руб.", "👨 Dima / Дима", "😐 neutral / нейтральный", True, True],
["Ты это серьёзно? Я даже не удивлена.", "👩 Masha / Маша", "😏 arrogance / высокомерие", True, True],
["Тише, дети уже спят, говори шёпотом.", "👩 Sveta / Света", "🤫 whisper / шёпот", True, True],
["Карл у Клары украл кораллы, а Клара у Карла украла кларнет.", "👨 Dima / Дима", "👅 tongue-twister / скороговорка", True, True],
]
def generate(text, speaker_label, emotion_label, normalize, auto_stress, speed, expressiveness):
speaker_id = SPEAKER_LABEL_TO_ID.get(speaker_label, 0)
emotion_id = EMOTION_LABEL_TO_ID.get(emotion_label, 0)
length_scale = 1.0 / max(0.1, float(speed))
try:
sr, audio, used = TTS.synthesize(
text,
speaker_id=speaker_id,
emotion_id=emotion_id,
normalize=bool(normalize),
auto_stress=bool(auto_stress),
noise_scale=float(expressiveness),
length_scale=length_scale,
)
except ValueError as e:
raise gr.Error(str(e))
return (sr, audio), used
with gr.Blocks(title="Dialogs-RU · Expressive Russian TTS", theme=gr.themes.Default()) as demo:
gr.Markdown(DESCRIPTION)
with gr.Row():
with gr.Column(scale=3):
text = gr.Textbox(
label="Russian text / Текст",
value="Привет! Это демо выразительного русского синтеза речи.",
lines=3,
placeholder="Введите текст на русском…",
)
with gr.Row():
speaker = gr.Dropdown(
SPEAKER_CHOICES, value=SPEAKER_CHOICES[0], label="Voice / Голос"
)
emotion = gr.Dropdown(
EMOTION_CHOICES, value=EMOTION_CHOICES[0], label="Emotion / Эмоция"
)
normalize = gr.Checkbox(
value=True,
label="Normalize numbers, dates & abbreviations · нормализация (rutextnorm)",
)
auto_stress = gr.Checkbox(
value=True, label="Auto stress (ruaccent) · авто-ударение"
)
with gr.Accordion("Advanced", open=False):
speed = gr.Slider(0.5, 1.8, value=1.0, step=0.05, label="Speed / Скорость")
expressiveness = gr.Slider(
0.0, 1.0, value=0.667, step=0.01,
label="Expressiveness (noise_scale) / Выразительность",
)
btn = gr.Button("🔊 Generate / Озвучить", variant="primary")
with gr.Column(scale=2):
audio_out = gr.Audio(label="Output / Результат", type="numpy")
used_text = gr.Textbox(
label="Text sent to the model (normalized + stressed)", interactive=False
)
gr.Markdown(TIPS)
btn.click(
generate,
inputs=[text, speaker, emotion, normalize, auto_stress, speed, expressiveness],
outputs=[audio_out, used_text],
)
gr.Examples(
examples=EXAMPLES,
inputs=[text, speaker, emotion, normalize, auto_stress],
label="Examples / Примеры",
)
gr.Markdown(
"Model & demo released under **OpenRAIL** · "
"trained on the [Dialogs](https://huggingface.co/datasets/langswap/dialogs-ru-emotional-conversations) corpus · "
"[training code](https://github.com/shigabeev/vits2-emotional)"
)
if __name__ == "__main__":
demo.queue(max_size=20).launch(server_name="0.0.0.0", server_port=7860)