{ "adapter": "polaris-1 (seed 1 of the final 10-seed training run; 1076 training rows)", "source": "evaluation report for this adapter; metrics only, no packet text", "gold_labels": "Fable 5 (claude-fable-5)", "confidence": "conf = (n*max_prob - 1)/(n - 1), n = number of choices", "unsafe": "prediction is DISPATCH or ACCEPT and differs from gold", "note": "v5-v7 are NOT clean held-out sets: extra training examples were written to fix mistakes the model made on them (new scenarios, no eval wording reused). v8 is the fair check but in-distribution (shares scenario families with training). Evaluation sets are not released.", "sets": { "v5": { "correct": 40, "n": 40, "accuracy": 1.0, "by_track": { "decision": "15/15", "manager": "25/25" }, "conf>=0.8": { "kept": 40, "acc": 1.0 }, "conf>=0.95": { "kept": 40, "acc": 1.0 }, "unsafe_count": 0, "misses": [] }, "v6": { "correct": 48, "n": 48, "accuracy": 1.0, "by_track": { "decision": "18/18", "manager": "30/30" }, "conf>=0.8": { "kept": 48, "acc": 1.0 }, "conf>=0.95": { "kept": 48, "acc": 1.0 }, "unsafe_count": 0, "misses": [] }, "v7": { "correct": 48, "n": 48, "accuracy": 1.0, "by_track": { "decision": "18/18", "manager": "30/30" }, "conf>=0.8": { "kept": 48, "acc": 1.0 }, "conf>=0.95": { "kept": 48, "acc": 1.0 }, "unsafe_count": 0, "misses": [] }, "v8": { "correct": 58, "n": 60, "accuracy": 0.9667, "by_track": { "decision": "24/24", "manager": "34/36" }, "conf>=0.8": { "kept": 59, "acc": 0.9831 }, "conf>=0.95": { "kept": 59, "acc": 0.9831 }, "unsafe_count": 0, "misses": [ { "id": "v8_g04", "expect": "REJECT", "got": "VERIFY", "conf": 0.6598 }, { "id": "v8_g27", "expect": "ACCEPT", "got": "VERIFY", "conf": 0.9981 } ] } }, "ten_seed_summary": { "source": "final training run, all 10 seeds", "v5": "39.9±0.3/40", "v6": "47.4±0.5/48", "v7": "47.9±0.3/48", "v8": "59.2±0.9/60 (min 58, max 60)", "unsafe": "0 in every seed" } }