pearlyjam21's picture
Reachy Mini multimodal emotion app
d626f8b verified
Raw History Blame
2.72 kB
"""End to end through MultimodalPipeline with the real models (skipped without MM_EMOTION_LOCAL_DIR)."""
import os
import time
import wave
from pathlib import Path
import numpy as np
import pytest
from reachy_mini_multimodal_emotion.engine.speech import SAMPLE_RATE
@pytest.fixture(scope="module")
def pipe(model_root):
from reachy_mini_multimodal_emotion.engine.pipeline import MultimodalPipeline
asr = os.environ.get("MM_EMOTION_ASR_DIR")
p = MultimodalPipeline(model_root, asr if asr and Path(asr).exists() else None)
yield p
p.close()
def voiced(seconds=2.0):
t = np.arange(int(SAMPLE_RATE * seconds)) / SAMPLE_RATE
return (0.08 * np.sign(np.sin(2 * np.pi * 150 * t)) * (0.6 + 0.4 * np.sin(2 * np.pi * 3 * t))).astype(np.float32)
def test_silence_and_no_face_give_no_emotion(pipe):
pipe.reset_audio()
pipe.push_audio(np.zeros(SAMPLE_RATE * 2, np.float32))
pipe.push_frame(np.full((480, 640, 3), 127, np.uint8))
snap = pipe.step()
assert snap.branches["speech"].status == "silence"
assert snap.branches["face"].status == "no_face"
assert snap.fused.probs is None
def test_voice_only_fusion_uses_speech_alone(pipe):
pipe.reset_audio()
pipe.push_audio(voiced())
pipe.push_frame(None)
snap = pipe.step()
assert snap.branches["speech"].present and snap.fused.used == {"speech": 1.0}
np.testing.assert_allclose(snap.fused.probs, snap.branches["speech"].probs, atol=1e-9)
def test_asr_utterance_reaches_text_branch(pipe):
if pipe.transcriber is None:
pytest.skip("set MM_EMOTION_ASR_DIR to the SenseVoice folder")
clip = Path(os.environ["MM_EMOTION_ASR_DIR"]) / "test_wavs" / "zh.wav"
if not clip.exists():
pytest.skip("SenseVoice zh.wav test clip not available")
with wave.open(str(clip)) as w:
audio = np.frombuffer(w.readframes(w.getnframes()), np.int16).astype(np.float32) / 32768
audio = np.concatenate([audio, np.zeros(SAMPLE_RATE, np.float32)]) # trailing silence ends the utterance
time.sleep(1.0) # let the worker finish segments left over from earlier tests
pipe.reset_audio()
pipe.utterances.clear()
for i in range(0, len(audio), 1024):
pipe.push_audio(audio[i:i + 1024])
deadline = time.monotonic() + 20
while not any(u["lang"] == "zh" for u in pipe.utterances) and time.monotonic() < deadline:
time.sleep(0.1)
zh = [u for u in pipe.utterances if u["lang"] == "zh"]
assert zh, f"no Mandarin utterance transcribed: {list(pipe.utterances)}"
assert "ๆ™‚้–“" in zh[0]["text"] # converted to Traditional
assert pipe.step().branches["text"].present