#!/usr/bin/env python3 import sys import os # Ensure the conda environment's bin and base Anaconda bin are in PATH so ffmpeg can be found bin_dir = os.path.dirname(sys.executable) anaconda_bin = "/home/jason/anaconda3/bin" paths_to_add = [bin_dir, anaconda_bin] current_path = os.environ.get("PATH", "") path_list = current_path.split(os.path.pathsep) if current_path else [] for p in paths_to_add: if p and p not in path_list: path_list.insert(0, p) os.environ["PATH"] = os.path.pathsep.join(path_list) import json import re import urllib.request import subprocess import threading import queue import time import shutil import tempfile # --- ANSI Terminal Colors --- C_BLUE = "\033[94m" C_GREEN = "\033[92m" C_YELLOW = "\033[93m" C_MAGENTA = "\033[95m" C_CYAN = "\033[96m" C_BOLD = "\033[1m" C_RESET = "\033[0m" # --- Configuration --- OLLAMA_API_URL = "http://127.0.0.1:11434/api/chat" MODEL_NAME = "jason_twin:latest" RECORD_FILE = "/tmp/chat_twin_record.wav" # Speech/TTS/STT Configuration tts_config = { "voice": "cloned", # Options: cloned, male1, male2, male3, female1, female2, female3, child_male, child_female "rate": "0", # -100 to 100 "pitch": "0", # -100 to 100 "volume": "0", # -100 to 100 "enabled": True, "fallback_espeak": False } # Voice Cloning & Cloned Speaker Paths (/local-tts) CLONED_VOICE_CANDIDATES = [ "/home/jason/local-tts/cloned_output.wav", "/home/jason/local-tts/my_voice_clean.wav", os.path.join(os.path.dirname(os.path.abspath(__file__)), "cloned_output.wav"), os.path.join(os.path.dirname(os.path.abspath(__file__)), "my_voice_clean.wav"), ] def get_cloned_voice_path(): """Locate the cloned speaker reference audio file.""" for p in CLONED_VOICE_CANDIDATES: if os.path.exists(p) and os.path.getsize(p) > 1000: return p return None TTS_CMD_CANDIDATES = [ "/home/jason/miniconda3/envs/env_twin/bin/tts", "/home/jason/.local/bin/tts", "/home/jason/local-tts/tts-env/bin/tts", "/home/jason/local-tts/bin/tts", ] def get_tts_cmd(): """Locate the TTS binary for voice cloning synthesis.""" for c in TTS_CMD_CANDIDATES: if c and os.path.exists(c): return c return shutil.which("tts") # --- Lazy-Loaded Speech-to-Text (STT) --- asr_pipeline = None def get_asr_pipeline(): """Lazily loads Whisper model only when voice mode is activated.""" global asr_pipeline if asr_pipeline is None: print(f"\n{C_YELLOW}[Initializing local Whisper Speech Recognition (GPU)...]{C_RESET}") from transformers import pipeline import logging logging.getLogger("transformers").setLevel(logging.ERROR) # Load a fast, local English model on the GPU (device 0) asr_pipeline = pipeline( "automatic-speech-recognition", model="openai/whisper-tiny.en", device=0 ) print(f"{C_GREEN}[Speech Recognition Ready!]{C_RESET}") return asr_pipeline # --- Thread-Safe Queue and Control for TTS --- tts_queue = queue.Queue() tts_stop_event = threading.Event() tts_thread = None def clean_markdown(text): """Clean markdown formatting from text for natural speech synthesis.""" # Remove code blocks text = re.sub(r'```.*?```', '', text, flags=re.DOTALL) # Remove inline code text = re.sub(r'`[^`]+`', '', text) # Remove bold/italic markup text = re.sub(r'\*\*|__|\*|_', '', text) # Remove markdown link syntax [text](url) -> text text = re.sub(r'\[([^\]]+)\]\([^\)]+\)', r'\1', text) # Remove headers text = re.sub(r'^#+\s+', '', text, flags=re.MULTILINE) # Remove list indicators text = re.sub(r'^-\s+', '', text, flags=re.MULTILINE) # Clean up double spacing and strip text = re.sub(r'\s+', ' ', text) return text.strip() def speak_text(text): """Executes the TTS system commands to say the text using cloned voice.""" if not tts_config["enabled"]: return clean_text = clean_markdown(text) if not clean_text: return # 1. Attempt Voice-Cloned TTS using XTTS v2 and cloned audio from /local-tts if tts_config["voice"] == "cloned" and not tts_config.get("fallback_espeak", False): cloned_wav = get_cloned_voice_path() tts_cmd = get_tts_cmd() if tts_cmd and cloned_wav: tmp_wav = None try: with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as tmp_file: tmp_wav = tmp_file.name cmd = [ tts_cmd, "--model_name", "tts_models/multilingual/multi-dataset/xtts_v2", "--text", clean_text, "--speaker_wav", cloned_wav, "--language_idx", "en", "--out_path", tmp_wav, "--use_cuda", "true" ] env = os.environ.copy() env["COQUI_TOS_AGREED"] = "1" res = subprocess.run(cmd, env=env, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, timeout=45) if res.returncode == 0 and os.path.exists(tmp_wav) and os.path.getsize(tmp_wav) > 1000: play_res = subprocess.run(["paplay", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) if play_res.returncode != 0: subprocess.run(["aplay", "-q", tmp_wav], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) try: os.remove(tmp_wav) except Exception: pass return except Exception: pass finally: if tmp_wav and os.path.exists(tmp_wav): try: os.remove(tmp_wav) except Exception: pass # 2. Prepare command for spd-say (standard voice fallback) if not tts_config["fallback_espeak"]: cmd = ["spd-say", "-w"] # -w waits until speaking is finished voice_target = tts_config["voice"] if tts_config["voice"] != "cloned" else "male1" if voice_target: cmd.extend(["-t", voice_target]) if tts_config["rate"]: cmd.extend(["-r", tts_config["rate"]]) if tts_config["pitch"]: cmd.extend(["-p", tts_config["pitch"]]) if tts_config["volume"]: cmd.extend(["-i", tts_config["volume"]]) cmd.append(clean_text) try: res = subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) if res.returncode == 0: return except Exception: # Fallback to espeak-ng if spd-say fails tts_config["fallback_espeak"] = True speak_text(text) return else: # 3. Fallback to espeak-ng cmd = ["espeak-ng"] try: rate_val = int(tts_config["rate"]) wpm = 175 + int(rate_val * 0.8) cmd.extend(["-s", str(wpm)]) except Exception: pass cmd.append(clean_text) try: subprocess.run(cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) except Exception as e: print(f"\n{C_YELLOW}[TTS Error: {e}]{C_RESET}") def stop_speech(): """Cancels any ongoing speech, terminates playback processes, and flushes the queue.""" # Clear the queue while not tts_queue.empty(): try: tts_queue.get_nowait() tts_queue.task_done() except queue.Empty: break # Kill any audio players or ongoing tts synthesis processes for proc in ["paplay", "aplay", "tts", "espeak-ng"]: try: subprocess.run(["pkill", "-x", proc], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) except Exception: pass # Send cancel to speech dispatcher try: subprocess.run(["spd-say", "-C"], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL) except Exception: pass def tts_worker(): """Worker thread that processes the speech queue sequentially.""" while not tts_stop_event.is_set(): try: item = tts_queue.get(timeout=0.5) if item is None: break speak_text(item) tts_queue.task_done() except queue.Empty: continue except Exception as e: print(f"\n{C_YELLOW}[TTS Worker Error: {e}]{C_RESET}") def start_tts_system(): global tts_thread, tts_stop_event tts_stop_event.clear() tts_thread = threading.Thread(target=tts_worker, daemon=True) tts_thread.start() def shutdown_tts_system(): tts_stop_event.set() tts_queue.put(None) if tts_thread: tts_thread.join(timeout=1.0) # --- Voice Recording and Transcription --- def record_audio(filename=RECORD_FILE): """Starts background audio recording via arecord.""" if os.path.exists(filename): try: os.remove(filename) except: pass # Record mono at 16kHz (best format for Whisper input) process = subprocess.Popen( ["arecord", "-f", "S16_LE", "-c", "1", "-r", "16000", "-t", "wav", filename], stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL ) return process def transcribe_audio(filename=RECORD_FILE): """Transcribes the recorded audio file using Whisper.""" try: pipe = get_asr_pipeline() except Exception as e: print(f"\n{C_YELLOW}Failed to load Whisper STT: {e}{C_RESET}") return "" if not os.path.exists(filename) or os.path.getsize(filename) < 100: return "" try: result = pipe(filename) return result.get("text", "").strip() except Exception as e: print(f"\n{C_YELLOW}Error during transcription: {e}{C_RESET}") return "" # --- Chat Logic --- def print_help(): print(f"\n{C_MAGENTA}{C_BOLD}--- Digital Twin Help Menu ---{C_RESET}") print(f" {C_BOLD}/stop{C_RESET} or {C_BOLD}/s{C_RESET} : Stop current speaking immediately.") print(f" {C_BOLD}/voice {C_RESET} : Set voice (cloned [from /local-tts], male1, male2, male3, female1, female2, female3, child_male, child_female).") print(f" {C_BOLD}/rate {C_RESET} : Set speech rate (-100 to 100, e.g. -20, 0, 20).") print(f" {C_BOLD}/clear{C_RESET} or {C_BOLD}/c{C_RESET} : Clear screen and reset conversation history.") print(f" {C_BOLD}/tts {C_RESET} : Enable or disable Text-to-Speech.") print(f" {C_BOLD}/help{C_RESET} or {C_BOLD}/h{C_RESET} : Show this help menu.") print(f" {C_BOLD}/exit{C_RESET} or {C_BOLD}/quit{C_RESET} : Exit the digital twin.") print(f" {C_BOLD}[Empty Prompt]{C_RESET} : Just press Enter without typing to trigger {C_BOLD}Voice Input Mode{C_RESET}.") print(f"{C_MAGENTA}------------------------------{C_RESET}\n") def check_ollama(): """Checks if Ollama is running and has the model.""" try: req = urllib.request.Request("http://127.0.0.1:11434/") with urllib.request.urlopen(req, timeout=2.0) as response: return response.status == 200 except Exception: return False def stream_response(messages): """Sends a chat request to Ollama and streams the response token by token.""" data = { "model": MODEL_NAME, "messages": messages, "stream": True } req = urllib.request.Request( OLLAMA_API_URL, data=json.dumps(data).encode("utf-8"), headers={"Content-Type": "application/json"} ) try: response = urllib.request.urlopen(req) return response except Exception as e: print(f"\n{C_YELLOW}Failed to connect to Ollama: {e}{C_RESET}") return None def main(): print(f"{C_CYAN}{C_BOLD}") print("====================================================") print(" JASON'S AI DIGITAL TWIN (2-WAY VOICE CHAT) ") print("====================================================") print(f"{C_RESET}") print("Connecting to local Ollama server...") if not check_ollama(): print(f"\n{C_YELLOW}Error: Ollama is not running!{C_RESET}") print("Please start Ollama (e.g. running 'ollama serve' or open Ollama app) and make sure it has 'jason_twin:latest' loaded.") input("\nPress Enter to exit...") return print(f"Twin model '{C_BOLD}{MODEL_NAME}{C_RESET}' loaded.") cloned_path = get_cloned_voice_path() if cloned_path and tts_config["voice"] == "cloned": print(f"Voice Profile: {C_BOLD}{C_GREEN}Jason's Cloned Voice ({cloned_path}){C_RESET}") else: print(f"Voice Profile: {C_BOLD}{C_MAGENTA}{tts_config['voice']}{C_RESET}") print(f"Two-Way Voice active. Type text OR press {C_BOLD}[ENTER]{C_RESET} to speak. (Type {C_BOLD}/help{C_RESET} for commands)\n") start_tts_system() chat_history = [] while True: try: # Display prompt prompt_label = f"\n{C_CYAN}{C_BOLD}You [Press ENTER for Voice, or type]:{C_RESET} " user_input = input(prompt_label).strip() # Stop any ongoing speech when user starts interacting stop_speech() # If input is empty, enter Voice Input Mode if not user_input: print(f"{C_YELLOW}[Recording... Press ENTER to stop recording]{C_RESET}", end="", flush=True) # Start recording record_proc = record_audio() # Wait for user to press enter to stop recording try: input() except (KeyboardInterrupt, EOFError): # Handle cancel recording record_proc.terminate() record_proc.wait() print(f"\n{C_MAGENTA}Recording cancelled.{C_RESET}") continue # Stop recording record_proc.terminate() record_proc.wait() print(f"{C_YELLOW}[Transcribing audio...]{C_RESET}", end="", flush=True) transcription = transcribe_audio() # Clear the "[Transcribing...]" status line sys.stdout.write("\r" + " " * 50 + "\r") sys.stdout.flush() if not transcription: print(f"{C_YELLOW}No speech recognized. Please try again.{C_RESET}") continue print(f"{C_CYAN}{C_BOLD}You (Voice):{C_RESET} {transcription}") user_input = transcription # Process slash commands if user_input.startswith("/"): cmd_parts = user_input.split() cmd = cmd_parts[0].lower() if cmd in ["/exit", "/quit"]: break elif cmd in ["/stop", "/s"]: print(f"{C_MAGENTA}Speech silenced.{C_RESET}") continue elif cmd in ["/help", "/h"]: print_help() continue elif cmd in ["/clear", "/c"]: chat_history = [] os.system('clear' if os.name == 'posix' else 'cls') print(f"{C_CYAN}{C_BOLD}===================================================={C_RESET}") print(f"{C_CYAN}{C_BOLD} JASON'S AI DIGITAL TWIN (2-WAY VOICE CHAT) {C_RESET}") print(f"{C_CYAN}{C_BOLD}===================================================={C_RESET}") print("Conversation history reset.") continue elif cmd == "/voice": if len(cmd_parts) > 1: new_voice = cmd_parts[1].lower() valid_voices = ["cloned", "male1", "male2", "male3", "female1", "female2", "female3", "child_male", "child_female"] if new_voice in valid_voices: tts_config["voice"] = new_voice if new_voice == "cloned": p = get_cloned_voice_path() print(f"{C_MAGENTA}Voice updated to: Jason's Cloned Voice ({p}){C_RESET}") else: print(f"{C_MAGENTA}Voice updated to: {new_voice}{C_RESET}") else: print(f"{C_YELLOW}Invalid voice. Select from: {', '.join(valid_voices)}{C_RESET}") else: if tts_config['voice'] == 'cloned': p = get_cloned_voice_path() print(f"{C_MAGENTA}Current voice: Jason's Cloned Voice ({p}){C_RESET}") else: print(f"{C_MAGENTA}Current voice: {tts_config['voice']}{C_RESET}") continue elif cmd == "/rate": if len(cmd_parts) > 1: try: rate_val = int(cmd_parts[1]) if -100 <= rate_val <= 100: tts_config["rate"] = str(rate_val) print(f"{C_MAGENTA}Speech rate updated to: {rate_val}{C_RESET}") else: print(f"{C_YELLOW}Speech rate must be between -100 and 100.{C_RESET}") except ValueError: print(f"{C_YELLOW}Rate must be an integer.{C_RESET}") else: print(f"{C_MAGENTA}Current speech rate: {tts_config['rate']}{C_RESET}") continue elif cmd == "/tts": if len(cmd_parts) > 1: state = cmd_parts[1].lower() if state in ["on", "enable", "true"]: tts_config["enabled"] = True print(f"{C_MAGENTA}Text-to-Speech enabled.{C_RESET}") elif state in ["off", "disable", "false"]: tts_config["enabled"] = False print(f"{C_MAGENTA}Text-to-Speech disabled.{C_RESET}") else: print(f"{C_YELLOW}Use '/tts on' or '/tts off'{C_RESET}") else: status = "enabled" if tts_config["enabled"] else "disabled" print(f"{C_MAGENTA}TTS is currently {status}.{C_RESET}") continue else: print(f"{C_YELLOW}Unknown command. Type /help to see available commands.{C_RESET}") continue # Standard user chat message chat_history.append({"role": "user", "content": user_input}) # Print response header print(f"\n{C_GREEN}{C_BOLD}AI Twin:{C_RESET} ", end="", flush=True) # Start streaming response stream = stream_response(chat_history) if not stream: chat_history.pop() continue full_response = "" sentence_buffer = "" for line in stream: if line: chunk = json.loads(line.decode("utf-8")) message = chunk.get("message", {}) token = message.get("content", "") if token: print(token, end="", flush=True) full_response += token sentence_buffer += token # Check for sentence boundaries to push to TTS queue while True: match = re.search(r'([.?!])\s+|\n', sentence_buffer) if not match: break end_pos = match.end() sentence = sentence_buffer[:end_pos].strip() sentence_buffer = sentence_buffer[end_pos:] if sentence and tts_config["enabled"]: if not sentence.startswith("```"): tts_queue.put(sentence) # Enqueue any remaining text in the buffer remaining = sentence_buffer.strip() if remaining and tts_config["enabled"] and not remaining.startswith("```"): tts_queue.put(remaining) print() # Print final newline chat_history.append({"role": "assistant", "content": full_response}) # Keep history under a sliding window to prevent context overflow if len(chat_history) > 13: chat_history = [chat_history[0]] + chat_history[-12:] except KeyboardInterrupt: # Handle Ctrl+C: stop speech, or exit if already stopped if not tts_queue.empty() or subprocess.run(["pgrep", "-x", "spd-say"], stdout=subprocess.DEVNULL).returncode == 0: print(f"\n{C_MAGENTA}Stopping speech...{C_RESET}") stop_speech() else: print("\nExiting...") break except Exception as e: print(f"\n{C_YELLOW}An error occurred: {e}{C_RESET}") shutdown_tts_system() # Clean up recording file if it exists if os.path.exists(RECORD_FILE): try: os.remove(RECORD_FILE) except: pass print(f"\n{C_CYAN}Goodbye!{C_RESET}") if __name__ == "__main__": main()