jlind456 commited on
Commit
aff52a3
·
verified ·
1 Parent(s): 24e8996

Update AI Twin chat client to use cloned voice from /local-tts

Browse files
Files changed (1) hide show
  1. chat_twin.py +110 -54
chat_twin.py CHANGED
@@ -1,4 +1,4 @@
1
- #!/home/jason/anaconda3/envs/env_twin/bin/python3
2
  import sys
3
  import os
4
 
@@ -20,6 +20,7 @@ import subprocess
20
  import threading
21
  import queue
22
  import time
 
23
  import tempfile
24
 
25
  # --- ANSI Terminal Colors ---
@@ -38,7 +39,7 @@ RECORD_FILE = "/tmp/chat_twin_record.wav"
38
 
39
  # Speech/TTS/STT Configuration
40
  tts_config = {
41
- "voice": "male1", # Options: male1, male2, male3, female1, female2, female3, child_male, child_female
42
  "rate": "0", # -100 to 100
43
  "pitch": "0", # -100 to 100
44
  "volume": "0", # -100 to 100
@@ -46,6 +47,36 @@ tts_config = {
46
  "fallback_espeak": False
47
  }
48
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
49
  # --- Lazy-Loaded Speech-to-Text (STT) ---
50
  asr_pipeline = None
51
 
@@ -90,9 +121,6 @@ def clean_markdown(text):
90
  text = re.sub(r'\s+', ' ', text)
91
  return text.strip()
92
 
93
- CLONED_SPEAKER_WAV = "/home/jason/local-tts/cloned_output.wav"
94
- TTS_CMD = "/home/jason/anaconda3/envs/tts-backend/bin/tts"
95
-
96
  def speak_text(text):
97
  """Executes the TTS system commands to say the text using cloned voice."""
98
  if not tts_config["enabled"]:
@@ -102,39 +130,52 @@ def speak_text(text):
102
  if not clean_text:
103
  return
104
 
105
- # Attempt voice-cloned TTS using XTTS v2 and cloned_output.wav
106
- if os.path.exists(TTS_CMD) and os.path.exists(CLONED_SPEAKER_WAV) and not tts_config.get("fallback_espeak", False):
107
- try:
108
- with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp_file:
109
- tmp_wav = tmp_file.name
110
-
111
- cmd = [
112
- TTS_CMD,
113
- "--model_name", "tts_models/multilingual/multi-dataset/xtts_v2",
114
- "--text", clean_text,
115
- "--speaker_wav", CLONED_SPEAKER_WAV,
116
- "--language_idx", "en",
117
- "--out_path", tmp_wav,
118
- "--use_cuda", "true"
119
- ]
120
- res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, timeout=45)
121
- if res.returncode == 0 and os.path.exists(tmp_wav):
122
- play_res = subprocess.run(["paplay", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
123
- if play_res.returncode != 0:
124
- subprocess.run(["aplay", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
125
- try:
126
- os.remove(tmp_wav)
127
- except Exception:
128
- pass
129
- return
130
- except Exception:
131
- pass
132
-
133
- # Prepare command for spd-say
 
 
 
 
 
 
 
 
 
 
 
 
134
  if not tts_config["fallback_espeak"]:
135
  cmd = ["spd-say", "-w"] # -w waits until speaking is finished
136
- if tts_config["voice"]:
137
- cmd.extend(["-t", tts_config["voice"]])
 
138
  if tts_config["rate"]:
139
  cmd.extend(["-r", tts_config["rate"]])
140
  if tts_config["pitch"]:
@@ -144,19 +185,22 @@ def speak_text(text):
144
  cmd.append(clean_text)
145
 
146
  try:
147
- subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
 
 
148
  except Exception:
149
  # Fallback to espeak-ng if spd-say fails
150
  tts_config["fallback_espeak"] = True
151
  speak_text(text)
 
152
  else:
153
- # Fallback to espeak-ng
154
  cmd = ["espeak-ng"]
155
  try:
156
  rate_val = int(tts_config["rate"])
157
  wpm = 175 + int(rate_val * 0.8)
158
  cmd.extend(["-s", str(wpm)])
159
- except:
160
  pass
161
  cmd.append(clean_text)
162
  try:
@@ -165,7 +209,7 @@ def speak_text(text):
165
  print(f"\n{C_YELLOW}[TTS Error: {e}]{C_RESET}")
166
 
167
  def stop_speech():
168
- """Cancels any ongoing speech and flushes the queue."""
169
  # Clear the queue
170
  while not tts_queue.empty():
171
  try:
@@ -174,16 +218,17 @@ def stop_speech():
174
  except queue.Empty:
175
  break
176
 
 
 
 
 
 
 
 
177
  # Send cancel to speech dispatcher
178
  try:
179
  subprocess.run(["spd-say", "-C"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
180
- except:
181
- pass
182
-
183
- # Terminate any running espeak-ng processes
184
- try:
185
- subprocess.run(["pkill", "-x", "espeak-ng"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
186
- except:
187
  pass
188
 
189
  def tts_worker():
@@ -254,7 +299,7 @@ def transcribe_audio(filename=RECORD_FILE):
254
  def print_help():
255
  print(f"\n{C_MAGENTA}{C_BOLD}--- Digital Twin Help Menu ---{C_RESET}")
256
  print(f" {C_BOLD}/stop{C_RESET} or {C_BOLD}/s{C_RESET} : Stop current speaking immediately.")
257
- print(f" {C_BOLD}/voice <type>{C_RESET} : Set voice (male1, male2, male3, female1, female2, female3, child_male, child_female).")
258
  print(f" {C_BOLD}/rate <value>{C_RESET} : Set speech rate (-100 to 100, e.g. -20, 0, 20).")
259
  print(f" {C_BOLD}/clear{C_RESET} or {C_BOLD}/c{C_RESET} : Clear screen and reset conversation history.")
260
  print(f" {C_BOLD}/tts <on|off>{C_RESET} : Enable or disable Text-to-Speech.")
@@ -308,6 +353,11 @@ def main():
308
  return
309
 
310
  print(f"Twin model '{C_BOLD}{MODEL_NAME}{C_RESET}' loaded.")
 
 
 
 
 
311
  print(f"Two-Way Voice active. Type text OR press {C_BOLD}[ENTER]{C_RESET} to speak. (Type {C_BOLD}/help{C_RESET} for commands)\n")
312
 
313
  start_tts_system()
@@ -325,9 +375,7 @@ def main():
325
 
326
  # If input is empty, enter Voice Input Mode
327
  if not user_input:
328
- if os.path.exists("/home/jason/coral/mic_active.wav"):
329
- subprocess.run(["aplay", "-q", "/home/jason/coral/mic_active.wav"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
330
- print(f"{C_YELLOW}[🎙️ Mic Active - Recording... Press ENTER to stop recording]{C_RESET}", end="", flush=True)
331
 
332
  # Start recording
333
  record_proc = record_audio()
@@ -384,14 +432,22 @@ def main():
384
  elif cmd == "/voice":
385
  if len(cmd_parts) > 1:
386
  new_voice = cmd_parts[1].lower()
387
- valid_voices = ["male1", "male2", "male3", "female1", "female2", "female3", "child_male", "child_female"]
388
  if new_voice in valid_voices:
389
  tts_config["voice"] = new_voice
390
- print(f"{C_MAGENTA}Voice updated to: {new_voice}{C_RESET}")
 
 
 
 
391
  else:
392
  print(f"{C_YELLOW}Invalid voice. Select from: {', '.join(valid_voices)}{C_RESET}")
393
  else:
394
- print(f"{C_MAGENTA}Current voice: {tts_config['voice']}{C_RESET}")
 
 
 
 
395
  continue
396
  elif cmd == "/rate":
397
  if len(cmd_parts) > 1:
 
1
+ #!/usr/bin/env python3
2
  import sys
3
  import os
4
 
 
20
  import threading
21
  import queue
22
  import time
23
+ import shutil
24
  import tempfile
25
 
26
  # --- ANSI Terminal Colors ---
 
39
 
40
  # Speech/TTS/STT Configuration
41
  tts_config = {
42
+ "voice": "cloned", # Options: cloned, male1, male2, male3, female1, female2, female3, child_male, child_female
43
  "rate": "0", # -100 to 100
44
  "pitch": "0", # -100 to 100
45
  "volume": "0", # -100 to 100
 
47
  "fallback_espeak": False
48
  }
49
 
50
+ # Voice Cloning & Cloned Speaker Paths (/local-tts)
51
+ CLONED_VOICE_CANDIDATES = [
52
+ "/home/jason/local-tts/cloned_output.wav",
53
+ "/home/jason/local-tts/my_voice_clean.wav",
54
+ os.path.join(os.path.dirname(os.path.abspath(__file__)), "cloned_output.wav"),
55
+ os.path.join(os.path.dirname(os.path.abspath(__file__)), "my_voice_clean.wav"),
56
+ ]
57
+
58
+ def get_cloned_voice_path():
59
+ """Locate the cloned speaker reference audio file."""
60
+ for p in CLONED_VOICE_CANDIDATES:
61
+ if os.path.exists(p) and os.path.getsize(p) > 1000:
62
+ return p
63
+ return None
64
+
65
+ TTS_CMD_CANDIDATES = [
66
+ "/home/jason/miniconda3/envs/env_twin/bin/tts",
67
+ "/home/jason/.local/bin/tts",
68
+ "/home/jason/local-tts/tts-env/bin/tts",
69
+ "/home/jason/local-tts/bin/tts",
70
+ ]
71
+
72
+ def get_tts_cmd():
73
+ """Locate the TTS binary for voice cloning synthesis."""
74
+ for c in TTS_CMD_CANDIDATES:
75
+ if c and os.path.exists(c):
76
+ return c
77
+ return shutil.which("tts")
78
+
79
+
80
  # --- Lazy-Loaded Speech-to-Text (STT) ---
81
  asr_pipeline = None
82
 
 
121
  text = re.sub(r'\s+', ' ', text)
122
  return text.strip()
123
 
 
 
 
124
  def speak_text(text):
125
  """Executes the TTS system commands to say the text using cloned voice."""
126
  if not tts_config["enabled"]:
 
130
  if not clean_text:
131
  return
132
 
133
+ # 1. Attempt Voice-Cloned TTS using XTTS v2 and cloned audio from /local-tts
134
+ if tts_config["voice"] == "cloned" and not tts_config.get("fallback_espeak", False):
135
+ cloned_wav = get_cloned_voice_path()
136
+ tts_cmd = get_tts_cmd()
137
+ if tts_cmd and cloned_wav:
138
+ tmp_wav = None
139
+ try:
140
+ with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp_file:
141
+ tmp_wav = tmp_file.name
142
+
143
+ cmd = [
144
+ tts_cmd,
145
+ "--model_name", "tts_models/multilingual/multi-dataset/xtts_v2",
146
+ "--text", clean_text,
147
+ "--speaker_wav", cloned_wav,
148
+ "--language_idx", "en",
149
+ "--out_path", tmp_wav,
150
+ "--use_cuda", "true"
151
+ ]
152
+ env = os.environ.copy()
153
+ env["COQUI_TOS_AGREED"] = "1"
154
+ res = subprocess.run(cmd, env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, timeout=45)
155
+ if res.returncode == 0 and os.path.exists(tmp_wav) and os.path.getsize(tmp_wav) > 1000:
156
+ play_res = subprocess.run(["paplay", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
157
+ if play_res.returncode != 0:
158
+ subprocess.run(["aplay", "-q", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
159
+ try:
160
+ os.remove(tmp_wav)
161
+ except Exception:
162
+ pass
163
+ return
164
+ except Exception:
165
+ pass
166
+ finally:
167
+ if tmp_wav and os.path.exists(tmp_wav):
168
+ try:
169
+ os.remove(tmp_wav)
170
+ except Exception:
171
+ pass
172
+
173
+ # 2. Prepare command for spd-say (standard voice fallback)
174
  if not tts_config["fallback_espeak"]:
175
  cmd = ["spd-say", "-w"] # -w waits until speaking is finished
176
+ voice_target = tts_config["voice"] if tts_config["voice"] != "cloned" else "male1"
177
+ if voice_target:
178
+ cmd.extend(["-t", voice_target])
179
  if tts_config["rate"]:
180
  cmd.extend(["-r", tts_config["rate"]])
181
  if tts_config["pitch"]:
 
185
  cmd.append(clean_text)
186
 
187
  try:
188
+ res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
189
+ if res.returncode == 0:
190
+ return
191
  except Exception:
192
  # Fallback to espeak-ng if spd-say fails
193
  tts_config["fallback_espeak"] = True
194
  speak_text(text)
195
+ return
196
  else:
197
+ # 3. Fallback to espeak-ng
198
  cmd = ["espeak-ng"]
199
  try:
200
  rate_val = int(tts_config["rate"])
201
  wpm = 175 + int(rate_val * 0.8)
202
  cmd.extend(["-s", str(wpm)])
203
+ except Exception:
204
  pass
205
  cmd.append(clean_text)
206
  try:
 
209
  print(f"\n{C_YELLOW}[TTS Error: {e}]{C_RESET}")
210
 
211
  def stop_speech():
212
+ """Cancels any ongoing speech, terminates playback processes, and flushes the queue."""
213
  # Clear the queue
214
  while not tts_queue.empty():
215
  try:
 
218
  except queue.Empty:
219
  break
220
 
221
+ # Kill any audio players or ongoing tts synthesis processes
222
+ for proc in ["paplay", "aplay", "tts", "espeak-ng"]:
223
+ try:
224
+ subprocess.run(["pkill", "-x", proc], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
225
+ except Exception:
226
+ pass
227
+
228
  # Send cancel to speech dispatcher
229
  try:
230
  subprocess.run(["spd-say", "-C"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
231
+ except Exception:
 
 
 
 
 
 
232
  pass
233
 
234
  def tts_worker():
 
299
  def print_help():
300
  print(f"\n{C_MAGENTA}{C_BOLD}--- Digital Twin Help Menu ---{C_RESET}")
301
  print(f" {C_BOLD}/stop{C_RESET} or {C_BOLD}/s{C_RESET} : Stop current speaking immediately.")
302
+ print(f" {C_BOLD}/voice <type>{C_RESET} : Set voice (cloned [from /local-tts], male1, male2, male3, female1, female2, female3, child_male, child_female).")
303
  print(f" {C_BOLD}/rate <value>{C_RESET} : Set speech rate (-100 to 100, e.g. -20, 0, 20).")
304
  print(f" {C_BOLD}/clear{C_RESET} or {C_BOLD}/c{C_RESET} : Clear screen and reset conversation history.")
305
  print(f" {C_BOLD}/tts <on|off>{C_RESET} : Enable or disable Text-to-Speech.")
 
353
  return
354
 
355
  print(f"Twin model '{C_BOLD}{MODEL_NAME}{C_RESET}' loaded.")
356
+ cloned_path = get_cloned_voice_path()
357
+ if cloned_path and tts_config["voice"] == "cloned":
358
+ print(f"Voice Profile: {C_BOLD}{C_GREEN}Jason's Cloned Voice ({cloned_path}){C_RESET}")
359
+ else:
360
+ print(f"Voice Profile: {C_BOLD}{C_MAGENTA}{tts_config['voice']}{C_RESET}")
361
  print(f"Two-Way Voice active. Type text OR press {C_BOLD}[ENTER]{C_RESET} to speak. (Type {C_BOLD}/help{C_RESET} for commands)\n")
362
 
363
  start_tts_system()
 
375
 
376
  # If input is empty, enter Voice Input Mode
377
  if not user_input:
378
+ print(f"{C_YELLOW}[Recording... Press ENTER to stop recording]{C_RESET}", end="", flush=True)
 
 
379
 
380
  # Start recording
381
  record_proc = record_audio()
 
432
  elif cmd == "/voice":
433
  if len(cmd_parts) > 1:
434
  new_voice = cmd_parts[1].lower()
435
+ valid_voices = ["cloned", "male1", "male2", "male3", "female1", "female2", "female3", "child_male", "child_female"]
436
  if new_voice in valid_voices:
437
  tts_config["voice"] = new_voice
438
+ if new_voice == "cloned":
439
+ p = get_cloned_voice_path()
440
+ print(f"{C_MAGENTA}Voice updated to: Jason's Cloned Voice ({p}){C_RESET}")
441
+ else:
442
+ print(f"{C_MAGENTA}Voice updated to: {new_voice}{C_RESET}")
443
  else:
444
  print(f"{C_YELLOW}Invalid voice. Select from: {', '.join(valid_voices)}{C_RESET}")
445
  else:
446
+ if tts_config['voice'] == 'cloned':
447
+ p = get_cloned_voice_path()
448
+ print(f"{C_MAGENTA}Current voice: Jason's Cloned Voice ({p}){C_RESET}")
449
+ else:
450
+ print(f"{C_MAGENTA}Current voice: {tts_config['voice']}{C_RESET}")
451
  continue
452
  elif cmd == "/rate":
453
  if len(cmd_parts) > 1: