#!/usr/bin/env python3 """GPQA Diamond against an OpenAI-compatible endpoint, following AA's published MCQ methodology (simple-evals style prompt, 'Answer: $LETTER' extraction, deterministically shuffled choices). Reference: GLM-5.2 original = 91.2.""" import argparse import asyncio import json import random import re import time from pathlib import Path from datasets import load_dataset from openai import AsyncOpenAI PROMPT_TMPL = ( "Answer the following multiple choice question. The last line of your response " "should be of the following format: 'Answer: $LETTER' (without quotes) where " "LETTER is one of ABCD. Think step by step before answering.\n\n" "{question}\n\nA) {a}\nB) {b}\nC) {c}\nD) {d}" ) ANSWER_RE = re.compile(r"(?i)answer\s*:\s*\*{0,2}\$?\(?\s*([A-D])\s*\)?") def extract_letter(content: str): hits = ANSWER_RE.findall(content) return hits[-1].upper() if hits else None async def run_one(client, sem, model, q, idx, rep, max_tokens, results, t0, total): rng = random.Random(idx * 1000 + rep) choices = [(q["Correct Answer"], True), (q["Incorrect Answer 1"], False), (q["Incorrect Answer 2"], False), (q["Incorrect Answer 3"], False)] rng.shuffle(choices) gold_letter = "ABCD"[[c[1] for c in choices].index(True)] prompt = PROMPT_TMPL.format(question=q["Question"], a=choices[0][0].strip(), b=choices[1][0].strip(), c=choices[2][0].strip(), d=choices[3][0].strip()) async with sem: last_err = None for attempt in range(4): try: start = time.monotonic() resp = await client.chat.completions.create( model=model, messages=[{"role": "user", "content": prompt}], temperature=1.0, top_p=0.95, max_tokens=max_tokens, ) secs = time.monotonic() - start content = resp.choices[0].message.content or "" pred = extract_letter(content) ok = pred == gold_letter rec = {"idx": idx, "repeat": rep, "gold": gold_letter, "pred": pred, "correct": ok, "finish_reason": resp.choices[0].finish_reason, "completion_tokens": resp.usage.completion_tokens if resp.usage else None, "secs": round(secs, 1), "content_tail": content[-300:]} results.append(rec) done = len(results) acc = sum(r["correct"] for r in results) / done print(f"[{done}/{total}] idx={idx} rep={rep} {'OK ' if ok else 'MISS'} " f"pred={pred} gold={gold_letter} tok={rec['completion_tokens']} {rec['secs']}s " f"| acc={acc:.3f} | {(time.monotonic()-t0)/60:.1f}m", flush=True) return except Exception as e: last_err = e wait = 15 * (attempt + 1) print(f"RETRY idx={idx} rep={rep} attempt={attempt+1}: {type(e).__name__}: {e}", flush=True) await asyncio.sleep(wait) results.append({"idx": idx, "repeat": rep, "gold": gold_letter, "pred": None, "correct": False, "finish_reason": f"error:{last_err}", "completion_tokens": None, "secs": None, "content_tail": ""}) async def main(): ap = argparse.ArgumentParser() ap.add_argument("--base-url", default="http://localhost:8000/v1") ap.add_argument("--api-key", default="dummy") ap.add_argument("--served-root", default="madeby561/GLM-5.2-MXFP8-NVFP4-NF3-Hybrid") ap.add_argument("--model", default="GLM-5.2") ap.add_argument("--repeats", type=int, default=2) ap.add_argument("--concurrency", type=int, default=6) ap.add_argument("--max-tokens", type=int, default=131072) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--out-dir", default="results") args = ap.parse_args() ds = load_dataset("Idavidrein/gpqa", "gpqa_diamond") items = list(ds[list(ds.keys())[0]]) if args.limit: items = items[: args.limit] out_dir = Path(args.out_dir) out_dir.mkdir(parents=True, exist_ok=True) client = AsyncOpenAI(base_url=args.base_url, api_key=args.api_key, timeout=14400.0, max_retries=0) sem = asyncio.Semaphore(args.concurrency) results = [] total = len(items) * args.repeats t0 = time.monotonic() print(f"=== GPQA Diamond: {len(items)} questions x {args.repeats} repeats = {total} generations ===", flush=True) await asyncio.gather(*[ run_one(client, sem, args.model, q, i, rep, args.max_tokens, results, t0, total) for rep in range(args.repeats) for i, q in enumerate(items) ]) with (out_dir / "gpqa_diamond_samples.jsonl").open("w") as f: for r in sorted(results, key=lambda r: (r["repeat"], r["idx"])): f.write(json.dumps(r) + "\n") n = len(results) toks = [r["completion_tokens"] for r in results if r["completion_tokens"]] summary = { "dataset": "Idavidrein/gpqa:gpqa_diamond", "model": args.model, "served_root": args.served_root, "settings": {"temperature": 1.0, "top_p": 0.95, "max_tokens": args.max_tokens, "prompt": "AA/simple-evals MCQ template", "grader": "regex letter match"}, "n_questions": len(items), "repeats": args.repeats, "accuracy_pass_at_1": round(sum(r["correct"] for r in results) / n, 4), "no_answer_extracted": sum(1 for r in results if r["pred"] is None), "truncated": sum(1 for r in results if r["finish_reason"] == "length"), "errors": sum(1 for r in results if str(r["finish_reason"]).startswith("error")), "avg_completion_tokens": round(sum(toks) / len(toks)) if toks else None, "wall_minutes": round((time.monotonic() - t0) / 60, 1), "reference_original_model": 91.2, } (out_dir / "gpqa_diamond_summary.json").write_text(json.dumps(summary, indent=2)) print(json.dumps(summary, indent=2), flush=True) if __name__ == "__main__": asyncio.run(main())