hoanglinhn0's picture
Update app.py
1c7e284
Raw
History Blame
4.54 kB
import gradio as gr
import onnxruntime as ort
import numpy as np
import json
import pysrt
import re
import os
from vinorm import TTSnorm
from phonemizer import phonemize
from scipy.io import wavfile
# --- 1. DANH SÁCH GIỌNG NÓI TỪ ẢNH GOOGLE DRIVE ---
# Bạn có thể thêm hoặc sửa tên hiển thị ở đây
VOICE_MAP = {
"Phương Mai 1": "phuongmai1.onnx",
"Ngọc Ngân 3701": "ngocngan3701.onnx",
"Lạc Phi": "lacphi.onnx",
"Ngọc Huyền": "ngochuyen.onnx",
"Duy Oryx 3175": "duyoryx3175.onnx",
# Thêm các giọng khác nếu bạn đã upload tệp lên
}
# Biến toàn cục lưu trữ session để tiết kiệm RAM
current_sess = None
current_phoneme_map = None
current_sample_rate = 22050
def load_voice_model(name):
global current_sess, current_phoneme_map, current_sample_rate
onnx_file = VOICE_MAP[name]
json_file = onnx_file + ".json"
if not os.path.exists(onnx_file) or not os.path.exists(json_file):
raise FileNotFoundError(f"Thiếu tệp {onnx_file} hoặc {json_file} trên Space!")
with open(json_file, "r", encoding="utf-8") as f:
cfg = json.load(f)
current_phoneme_map = cfg["phoneme_id_map"]
current_sample_rate = cfg.get("audio", {}).get("sample_rate", 22050)
current_sess = ort.InferenceSession(onnx_file, providers=['CPUExecutionProvider'])
# --- 2. XỬ LÝ ÂM VỊ & ĐỒNG BỘ SRT ---
def process_text(text):
text = re.sub(r'<[^>]*>', '', text) # Làm sạch HTML
text = TTSnorm(text.replace('\n', ' ').strip()) # Chuẩn hóa tiếng Việt
phonemes = phonemize(text, language='vi', backend='espeak', strip=True)
ids = [current_phoneme_map.get("^", [1])[0]]
for p in phonemes:
if p in current_phoneme_map:
ids.extend(current_phoneme_map[p])
ids.append(current_phoneme_map.get("_", [0])[0])
ids.append(current_phoneme_map.get("$", [2])[0])
return ids
def srt_to_audio_sync(voice_name, srt_file, speed=1.30):
if srt_file is None: return "Chưa chọn file SRT", None
try:
load_voice_model(voice_name) # Nạp model được chọn
subs = pysrt.open(srt_file.name, encoding='utf-8')
final_audio_stream = []
current_sample_idx = 0
for sub in subs:
# Đồng bộ mốc thời gian SRT
start_sample = int((sub.start.ordinal / 1000.0) * current_sample_rate)
if start_sample > current_sample_idx:
final_audio_stream.append(np.zeros(start_sample - current_sample_idx, dtype=np.float32))
current_sample_idx = start_sample
# Inference ONNX
ids = process_text(sub.text)
scales = np.array([0.667, 0.8, 1.0/speed], dtype=np.float32)
outputs = current_sess.run(None, {
"input": np.array([ids], dtype=np.int64),
"input_lengths": np.array([len(ids)], dtype=np.int64),
"scales": scales
})
# Làm phẳng mảng 4D
audio_segment = outputs[0].flatten().astype(np.float32)
final_audio_stream.append(audio_segment)
current_sample_idx += len(audio_segment)
output_path = "synced_voice.wav"
wavfile.write(output_path, current_sample_rate, np.concatenate(final_audio_stream))
return f"Hoàn thành lồng tiếng với giọng: {voice_name}", output_path
except Exception as e:
return f"Lỗi: {str(e)}", None
# --- 3. GIAO DIỆN CHỌN TÊN GIỌNG ---
with gr.Blocks(title="Multi-Voice SRT Dubbing") as demo:
gr.Markdown("# 🗣️ Hệ thống lồng tiếng 7 giọng Tiếng Việt")
with gr.Row():
with gr.Column():
# Menu chọn tên giọng như trong ảnh Drive của bạn
voice_select = gr.Dropdown(choices=list(VOICE_MAP.keys()), value="Phương Mai 1", label="Chọn giọng đọc")
file_input = gr.File(label="Tải file SRT từ SD Card", file_types=[".srt"])
speed_input = gr.Slider(0.5, 2.0, value=1.30, label="Tốc độ (Mặc định 1.30)")
btn = gr.Button("🚀 Chạy lồng tiếng", variant="primary")
with gr.Column():
status = gr.Textbox(label="Trạng thái")
audio_out = gr.Audio(label="Audio hoàn chỉnh (Khớp thời gian)")
btn.click(srt_to_audio_sync, [voice_select, file_input, speed_input], [status, audio_out])
demo.launch(server_name="0.0.0.0")