MiniMax-H3-FL2VA-Pruned-IQ1-GGUF / scripts /evaluate-quality-iq1.sh
MarxistLeninist's picture
Add files using upload-large-folder tool
d18b291 verified
Raw
History Blame Contribute Delete
5.26 kB
#!/usr/bin/env bash
set -Eeuo pipefail
ROOT=/opt/h3
MODELS="$ROOT/models"
QUALITY="$ROOT/quality"
LOGS="$ROOT/logs/quality"
SD_CLI="$ROOT/src/stable-diffusion.cpp/build/bin/sd-cli"
FFMPEG="$ROOT/bin/ffmpeg"
PYTHON=/opt/conda/bin/python
LLM="$MODELS/qwen3vl_32b_minimax_h3-Q2_K_M.gguf"
VIDEO_VAE="$MODELS/vae/minimax_h3_video_vae_fp16.safetensors"
AUDIO_VAE="$MODELS/vae/minimax_h3_audio_vae_fp32.safetensors"
DEFAULT_PROMPT='A red fox trots through crunchy snow; audible footsteps, soft wind, natural synchronized stereo sound.'
model="${1:?model path required}"
label="${2:?label required}"
profile="${3:-quick}"
prompt="${4:-$DEFAULT_PROMPT}"
seed="${5:-11}"
[[ -s "$model" ]] || { echo "model missing or empty: $model" >&2; exit 1; }
[[ "$seed" =~ ^[0-9]+$ ]] || { echo "seed must be a non-negative integer" >&2; exit 2; }
case "$profile" in
quick)
width=320; height=192; frames=22; steps=4
;;
full)
width=640; height=384; frames=39; steps=8
;;
site)
width=640; height=384; frames=39; steps=4
;;
*)
echo "profile must be quick, site, or full" >&2
exit 2
;;
esac
outdir="$QUALITY/eval/$label-$profile"
raw="$outdir/$label-$profile.raw.webm"
mp4="$outdir/$label-$profile.mp4"
log="$LOGS/eval-$label-$profile.log"
frames_dir="$outdir/frames"
mkdir -p "$outdir" "$frames_dir" "$LOGS"
rm -f "$frames_dir"/*.png "$outdir/audio.f32"
"$PYTHON" - "$outdir/run.json" "$model" "$label" "$profile" "$prompt" "$seed" "$width" "$height" "$frames" "$steps" <<'PY'
import json
import pathlib
import sys
keys = ("model", "label", "profile", "prompt", "seed", "width", "height", "frames", "steps")
values = sys.argv[2:]
record = dict(zip(keys, values))
for key in ("seed", "width", "height", "frames", "steps"):
record[key] = int(record[key])
pathlib.Path(sys.argv[1]).write_text(json.dumps(record, indent=2, sort_keys=True) + "\n")
PY
"$SD_CLI" \
--mode vid_gen \
--diffusion-model "$model" \
--llm "$LLM" \
--vae "$VIDEO_VAE" \
--audio-vae "$AUDIO_VAE" \
--prompt "$prompt" \
--width "$width" --height "$height" --video-frames "$frames" --fps 24 \
--steps "$steps" --cfg-scale 1.0 --backend te=cpu --diffusion-fa \
--rng cpu --seed "$seed" --output "$raw" --verbose >"$log" 2>&1
"$FFMPEG" -hide_banner -loglevel error -y -i "$raw" \
-map 0:v:0 -map 0:a:0 -c:v libx264 -preset veryfast -crf 20 -pix_fmt yuv420p \
-c:a aac -b:a 160k -ar 32000 -movflags +faststart "$mp4"
ffprobe -v error -show_entries stream=index,codec_type,codec_name,channels,sample_rate,width,height,r_frame_rate \
-of json "$mp4" >"$outdir/ffprobe.json"
"$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" "$frames_dir/%03d.png"
"$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" -map 0:a:0 -f f32le -acodec pcm_f32le "$outdir/audio.f32"
"$PYTHON" - "$frames_dir" "$outdir/audio.f32" "$outdir/metrics.json" <<'PY'
import json
import math
import pathlib
import sys
import numpy as np
from PIL import Image
frames_dir = pathlib.Path(sys.argv[1])
audio_path = pathlib.Path(sys.argv[2])
output_path = pathlib.Path(sys.argv[3])
frame_paths = sorted(frames_dir.glob("*.png"))
if not frame_paths:
raise SystemExit("no decoded frames")
spatial_sd = []
laplacian = []
phase_grid = []
frame_arrays = []
for path in frame_paths:
frame = np.asarray(Image.open(path).convert("L"), dtype=np.float32)
frame_arrays.append(frame)
sd = float(frame.std())
lap = np.abs(
-4 * frame[1:-1, 1:-1]
+ frame[:-2, 1:-1]
+ frame[2:, 1:-1]
+ frame[1:-1, :-2]
+ frame[1:-1, 2:]
)
phases = np.array(
[[frame[y::16, x::16].mean() for x in range(16)] for y in range(16)],
dtype=np.float32,
)
spatial_sd.append(sd)
laplacian.append(float(lap.mean()))
phase_grid.append(float(phases.std() / max(sd, 1e-8)))
temporal_mad = [
float(np.abs(frame_arrays[i] - frame_arrays[i - 1]).mean())
for i in range(1, len(frame_arrays))
]
audio = np.fromfile(audio_path, dtype=np.float32)
if audio.size == 0 or audio.size % 2:
raise SystemExit("invalid stereo float audio")
stereo = audio.reshape(-1, 2)
peak = float(np.abs(stereo).max())
rms = float(np.sqrt(np.mean(stereo * stereo)))
corr = float(np.corrcoef(stereo[:, 0], stereo[:, 1])[0, 1])
metrics = {
"frames": len(frame_paths),
"mean_spatial_luma_sd": float(np.mean(spatial_sd)),
"mean_abs_laplacian": float(np.mean(laplacian)),
"mean_phase_grid_16": float(np.mean(phase_grid)),
"mean_adjacent_frame_mad": float(np.mean(temporal_mad)),
"audio_peak": peak,
"audio_peak_dbfs": 20 * math.log10(max(peak, 1e-12)),
"audio_rms_dbfs": 20 * math.log10(max(rms, 1e-12)),
"audio_clipped_fraction": float(np.mean(np.abs(stereo) >= 0.999)),
"audio_stereo_correlation": corr,
}
output_path.write_text(json.dumps(metrics, indent=2, sort_keys=True) + "\n")
print(json.dumps(metrics, sort_keys=True))
PY
mid=$((frames / 2))
"$FFMPEG" -hide_banner -loglevel error -y -i "$mp4" \
-vf "select='eq(n,0)+eq(n,${mid})+eq(n,$((frames - 1)))',scale=640:384:flags=lanczos,tile=3x1:padding=4:margin=4" \
-frames:v 1 "$outdir/contact.png"
sha256sum "$mp4" "$outdir/contact.png" "$outdir/metrics.json" "$outdir/run.json" | tee "$outdir/SHA256SUMS"
cat "$outdir/metrics.json"