post-train evals: humaneval pass@1 final
Browse files
stage2_paper_faithful/eval_results/paper_faithful_humaneval_t1.json
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"pass_at_1": 0.06707317073170732,
|
| 3 |
+
"n_runs_pass": 11,
|
| 4 |
+
"n_runs_total": 164,
|
| 5 |
+
"n_tasks": 164,
|
| 6 |
+
"ckpt": null,
|
| 7 |
+
"model": "Qwen/Qwen3.5-0.8B-Base",
|
| 8 |
+
"surgery": "rtpurbo",
|
| 9 |
+
"surgery_state_dict": "checkpoints/rtpurbo_qwen3_5_0_8b_base/stage2_paper_faithful/state_dict.pt",
|
| 10 |
+
"manifest": null,
|
| 11 |
+
"source": "eval_results/paper_faithful_humaneval_t1_samples_eval_results.json"
|
| 12 |
+
}
|
stage2_paper_faithful/eval_results/paper_faithful_humaneval_t1_samples_eval_results.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|