#!/usr/bin/env python3 """Replicate Z.ai's AIME/HMMT eval methodology against an OpenAI-compatible endpoint. Methodology per zai-org/GLM-5.2 model card: temperature=1.0, top_p=0.95, max generation 163840 tokens, Explanation/Exact Answer/Confidence system prompt. Grading: math-verify symbolic match instead of an LLM judge. """ import argparse import asyncio import json import re import time from pathlib import Path from datasets import load_dataset from math_verify import parse as mv_parse, verify as mv_verify from openai import AsyncOpenAI SYSTEM_PROMPT = ( "Your response should be in the following format:\n" "Explanation: {your explanation for your final answer}\n" "Exact Answer: {your succinct, final answer}\n" "Confidence: {your confidence score between 0% and 100% for your answer}." ) BOXED_RE = re.compile(r"\\boxed\{([^{}]+(?:\{[^{}]*\}[^{}]*)*)\}") EXACT_RE = re.compile(r"Exact Answer:\s*(.+?)(?:\n|Confidence:|$)", re.IGNORECASE | re.DOTALL) def extract_answer(content: str): m = EXACT_RE.search(content) if m: ans = m.group(1).strip().rstrip(".").strip("$ ").strip() if ans: return ans boxed = BOXED_RE.findall(content) if boxed: return boxed[-1].strip() return None def grade(pred, gold) -> bool: if pred is None: return False gold_s, pred_s = str(gold).strip(), str(pred).strip() if pred_s == gold_s: return True try: g = mv_parse(f"${gold_s}$") p = mv_parse(f"${pred_s}$") if g and p and mv_verify(g, p): return True except Exception: pass return False async def run_one(client, sem, model, item, rep, max_tokens, results, t0, total): async with sem: last_err = None for attempt in range(4): try: start = time.monotonic() resp = await client.chat.completions.create( model=model, messages=[ {"role": "system", "content": SYSTEM_PROMPT}, {"role": "user", "content": item["problem"]}, ], temperature=1.0, top_p=0.95, max_tokens=max_tokens, ) secs = time.monotonic() - start msg = resp.choices[0].message content = msg.content or "" reasoning = getattr(msg, "reasoning", None) or getattr(msg, "reasoning_content", None) or "" pred = extract_answer(content) ok = grade(pred, item["answer"]) rec = { "problem_idx": item["problem_idx"], "repeat": rep, "gold": str(item["answer"]), "pred": pred, "correct": ok, "finish_reason": resp.choices[0].finish_reason, "completion_tokens": resp.usage.completion_tokens if resp.usage else None, "secs": round(secs, 1), "content": content, "reasoning_head": reasoning[:500], } results.append(rec) done = len(results) acc = sum(r["correct"] for r in results) / done elapsed = time.monotonic() - t0 print( f"[{done}/{total}] idx={item['problem_idx']} rep={rep} " f"{'OK ' if ok else 'MISS'} pred={pred!r} gold={item['answer']!r} " f"tok={rec['completion_tokens']} {rec['secs']}s | running acc={acc:.3f} | elapsed={elapsed/60:.1f}m", flush=True, ) return rec except Exception as e: last_err = e wait = 15 * (attempt + 1) print(f"RETRY idx={item['problem_idx']} rep={rep} attempt={attempt+1}: {type(e).__name__}: {e} (sleep {wait}s)", flush=True) await asyncio.sleep(wait) rec = {"problem_idx": item["problem_idx"], "repeat": rep, "gold": str(item["answer"]), "pred": None, "correct": False, "finish_reason": f"error:{last_err}", "completion_tokens": None, "secs": None, "content": "", "reasoning_head": ""} results.append(rec) return rec async def main(): ap = argparse.ArgumentParser() ap.add_argument("--dataset", required=True) ap.add_argument("--base-url", default="http://localhost:8000/v1") ap.add_argument("--api-key", default="dummy") ap.add_argument("--served-root", default="madeby561/GLM-5.2-MXFP8-NVFP4-NF3-Hybrid") ap.add_argument("--model", default="GLM-5.2") ap.add_argument("--repeats", type=int, default=4) ap.add_argument("--concurrency", type=int, default=8) ap.add_argument("--max-tokens", type=int, default=163840) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--out-dir", default=".") args = ap.parse_args() ds = load_dataset(args.dataset) split = list(ds.keys())[0] items = list(ds[split]) if args.limit: items = items[: args.limit] tag = args.dataset.split("/")[-1] out_dir = Path(args.out_dir) out_dir.mkdir(parents=True, exist_ok=True) samples_path = out_dir / f"{tag}_samples.jsonl" summary_path = out_dir / f"{tag}_summary.json" client = AsyncOpenAI(base_url=args.base_url, api_key=args.api_key, timeout=14400.0, max_retries=0) sem = asyncio.Semaphore(args.concurrency) results = [] total = len(items) * args.repeats t0 = time.monotonic() print(f"=== {args.dataset}: {len(items)} problems x {args.repeats} repeats = {total} generations ===", flush=True) tasks = [ run_one(client, sem, args.model, item, rep, args.max_tokens, results, t0, total) for rep in range(args.repeats) for item in items ] await asyncio.gather(*tasks) with samples_path.open("w") as f: for r in sorted(results, key=lambda r: (r["repeat"], str(r["problem_idx"]))): f.write(json.dumps(r) + "\n") n = len(results) acc = sum(r["correct"] for r in results) / n toks = [r["completion_tokens"] for r in results if r["completion_tokens"]] truncated = sum(1 for r in results if r["finish_reason"] == "length") errors = sum(1 for r in results if str(r["finish_reason"]).startswith("error")) per_q = {} for r in results: per_q.setdefault(str(r["problem_idx"]), []).append(r["correct"]) summary = { "dataset": args.dataset, "model": args.model, "served_root": args.served_root, "settings": {"temperature": 1.0, "top_p": 0.95, "max_tokens": args.max_tokens, "system_prompt": "zai-explanation-exact-answer-confidence", "grader": "math-verify"}, "n_problems": len(items), "repeats": args.repeats, "accuracy_pass_at_1": round(acc, 4), "truncated": truncated, "errors": errors, "avg_completion_tokens": round(sum(toks) / len(toks)) if toks else None, "wall_minutes": round((time.monotonic() - t0) / 60, 1), "per_question_correct_rate": {k: round(sum(v) / len(v), 3) for k, v in sorted(per_q.items(), key=lambda kv: kv[0])}, } summary_path.write_text(json.dumps(summary, indent=2)) print(json.dumps(summary, indent=2), flush=True) if __name__ == "__main__": asyncio.run(main())