#!/usr/bin/env bash set -Eeuo pipefail ROOT=/opt/h3 MODELS="$ROOT/models" QUALITY="$ROOT/quality" LOGS="$ROOT/logs/quality" SD_CLI="$ROOT/src/stable-diffusion.cpp/build/bin/sd-cli" FFMPEG="$ROOT/bin/ffmpeg" PYTHON=/opt/conda/bin/python LLM="$MODELS/qwen3vl_32b_minimax_h3-Q2_K_M.gguf" VIDEO_VAE="$MODELS/vae/minimax_h3_video_vae_fp16.safetensors" AUDIO_VAE="$MODELS/vae/minimax_h3_audio_vae_fp32.safetensors" DEFAULT_PROMPT='A red fox trots through crunchy snow; audible footsteps, soft wind, natural synchronized stereo sound.' model="${1:?model path required}" label="${2:?label required}" profile="${3:-quick}" prompt="${4:-$DEFAULT_PROMPT}" seed="${5:-11}" [[ -s "$model" ]] || { echo "model missing or empty: $model" >&2; exit 1; } [[ "$seed" =~ ^[0-9]+$ ]] || { echo "seed must be a non-negative integer" >&2; exit 2; } case "$profile" in quick) width=320; height=192; frames=22; steps=4 ;; full) width=640; height=384; frames=39; steps=8 ;; site) width=640; height=384; frames=39; steps=4 ;; *) echo "profile must be quick, site, or full" >&2 exit 2 ;; esac outdir="$QUALITY/eval/$label-$profile" raw="$outdir/$label-$profile.raw.webm" mp4="$outdir/$label-$profile.mp4" log="$LOGS/eval-$label-$profile.log" frames_dir="$outdir/frames" mkdir -p "$outdir" "$frames_dir" "$LOGS" rm -f "$frames_dir"/*.png "$outdir/audio.f32" "$PYTHON" - "$outdir/run.json" "$model" "$label" "$profile" "$prompt" "$seed" "$width" "$height" "$frames" "$steps" <<'PY' import json import pathlib import sys keys = ("model", "label", "profile", "prompt", "seed", "width", "height", "frames", "steps") values = sys.argv[2:] record = dict(zip(keys, values)) for key in ("seed", "width", "height", "frames", "steps"): record[key] = int(record[key]) pathlib.Path(sys.argv[1]).write_text(json.dumps(record, indent=2, sort_keys=True) + "\n") PY "$SD_CLI" \ --mode vid_gen \ --diffusion-model "$model" \ --llm "$LLM" \ --vae "$VIDEO_VAE" \ --audio-vae "$AUDIO_VAE" \ --prompt "$prompt" \ --width "$width" --height "$height" --video-frames "$frames" --fps 24 \ --steps "$steps" --cfg-scale 1.0 --backend te=cpu --diffusion-fa \ --rng cpu --seed "$seed" --output "$raw" --verbose >"$log" 2>&1 "$FFMPEG" -hide_banner -loglevel error -y -i "$raw" \ -map 0:v:0 -map 0:a:0 -c:v libx264 -preset veryfast -crf 20 -pix_fmt yuv420p \ -c:a aac -b:a 160k -ar 32000 -movflags +faststart "$mp4" ffprobe -v error -show_entries stream=index,codec_type,codec_name,channels,sample_rate,width,height,r_frame_rate \ -of json "$mp4" >"$outdir/ffprobe.json" "$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" "$frames_dir/%03d.png" "$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" -map 0:a:0 -f f32le -acodec pcm_f32le "$outdir/audio.f32" "$PYTHON" - "$frames_dir" "$outdir/audio.f32" "$outdir/metrics.json" <<'PY' import json import math import pathlib import sys import numpy as np from PIL import Image frames_dir = pathlib.Path(sys.argv[1]) audio_path = pathlib.Path(sys.argv[2]) output_path = pathlib.Path(sys.argv[3]) frame_paths = sorted(frames_dir.glob("*.png")) if not frame_paths: raise SystemExit("no decoded frames") spatial_sd = [] laplacian = [] phase_grid = [] frame_arrays = [] for path in frame_paths: frame = np.asarray(Image.open(path).convert("L"), dtype=np.float32) frame_arrays.append(frame) sd = float(frame.std()) lap = np.abs( -4 * frame[1:-1, 1:-1] + frame[:-2, 1:-1] + frame[2:, 1:-1] + frame[1:-1, :-2] + frame[1:-1, 2:] ) phases = np.array( [[frame[y::16, x::16].mean() for x in range(16)] for y in range(16)], dtype=np.float32, ) spatial_sd.append(sd) laplacian.append(float(lap.mean())) phase_grid.append(float(phases.std() / max(sd, 1e-8))) temporal_mad = [ float(np.abs(frame_arrays[i] - frame_arrays[i - 1]).mean()) for i in range(1, len(frame_arrays)) ] audio = np.fromfile(audio_path, dtype=np.float32) if audio.size == 0 or audio.size % 2: raise SystemExit("invalid stereo float audio") stereo = audio.reshape(-1, 2) peak = float(np.abs(stereo).max()) rms = float(np.sqrt(np.mean(stereo * stereo))) corr = float(np.corrcoef(stereo[:, 0], stereo[:, 1])[0, 1]) metrics = { "frames": len(frame_paths), "mean_spatial_luma_sd": float(np.mean(spatial_sd)), "mean_abs_laplacian": float(np.mean(laplacian)), "mean_phase_grid_16": float(np.mean(phase_grid)), "mean_adjacent_frame_mad": float(np.mean(temporal_mad)), "audio_peak": peak, "audio_peak_dbfs": 20 * math.log10(max(peak, 1e-12)), "audio_rms_dbfs": 20 * math.log10(max(rms, 1e-12)), "audio_clipped_fraction": float(np.mean(np.abs(stereo) >= 0.999)), "audio_stereo_correlation": corr, } output_path.write_text(json.dumps(metrics, indent=2, sort_keys=True) + "\n") print(json.dumps(metrics, sort_keys=True)) PY mid=$((frames / 2)) "$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" \ -vf "select='eq(n,0)+eq(n,${mid})+eq(n,$((frames - 1)))',scale=640:384:flags=lanczos,tile=3x1:padding=4:margin=4" \ -frames:v 1 "$outdir/contact.png" sha256sum "$mp4" "$outdir/contact.png" "$outdir/metrics.json" "$outdir/run.json" | tee "$outdir/SHA256SUMS" cat "$outdir/metrics.json"