"""End to end through MultimodalPipeline with the real models (skipped without MM_EMOTION_LOCAL_DIR).""" import os import time import wave from pathlib import Path import numpy as np import pytest from reachy_mini_multimodal_emotion.engine.speech import SAMPLE_RATE @pytest.fixture(scope="module") def pipe(model_root): from reachy_mini_multimodal_emotion.engine.pipeline import MultimodalPipeline asr = os.environ.get("MM_EMOTION_ASR_DIR") p = MultimodalPipeline(model_root, asr if asr and Path(asr).exists() else None) yield p p.close() def voiced(seconds=2.0): t = np.arange(int(SAMPLE_RATE * seconds)) / SAMPLE_RATE return (0.08 * np.sign(np.sin(2 * np.pi * 150 * t)) * (0.6 + 0.4 * np.sin(2 * np.pi * 3 * t))).astype(np.float32) def test_silence_and_no_face_give_no_emotion(pipe): pipe.reset_audio() pipe.push_audio(np.zeros(SAMPLE_RATE * 2, np.float32)) pipe.push_frame(np.full((480, 640, 3), 127, np.uint8)) snap = pipe.step() assert snap.branches["speech"].status == "silence" assert snap.branches["face"].status == "no_face" assert snap.fused.probs is None def test_voice_only_fusion_uses_speech_alone(pipe): pipe.reset_audio() pipe.push_audio(voiced()) pipe.push_frame(None) snap = pipe.step() assert snap.branches["speech"].present and snap.fused.used == {"speech": 1.0} np.testing.assert_allclose(snap.fused.probs, snap.branches["speech"].probs, atol=1e-9) def test_asr_utterance_reaches_text_branch(pipe): if pipe.transcriber is None: pytest.skip("set MM_EMOTION_ASR_DIR to the SenseVoice folder") clip = Path(os.environ["MM_EMOTION_ASR_DIR"]) / "test_wavs" / "zh.wav" if not clip.exists(): pytest.skip("SenseVoice zh.wav test clip not available") with wave.open(str(clip)) as w: audio = np.frombuffer(w.readframes(w.getnframes()), np.int16).astype(np.float32) / 32768 audio = np.concatenate([audio, np.zeros(SAMPLE_RATE, np.float32)]) # trailing silence ends the utterance time.sleep(1.0) # let the worker finish segments left over from earlier tests pipe.reset_audio() pipe.utterances.clear() for i in range(0, len(audio), 1024): pipe.push_audio(audio[i:i + 1024]) deadline = time.monotonic() + 20 while not any(u["lang"] == "zh" for u in pipe.utterances) and time.monotonic() < deadline: time.sleep(0.1) zh = [u for u in pipe.utterances if u["lang"] == "zh"] assert zh, f"no Mandarin utterance transcribed: {list(pipe.utterances)}" assert "時間" in zh[0]["text"] # converted to Traditional assert pipe.step().branches["text"].present