diff --git a/data/benchmarks.json b/data/benchmarks.json index 9571727dfcd0f5a5c4445684a0e339ea8b9100b5..90e9c7c4a82eb916eea7f40ad4a052dd6581fb2c 100644 --- a/data/benchmarks.json +++ b/data/benchmarks.json @@ -45,7 +45,7 @@ }, { "benchmark": "hfopenllm_v2", - "model_count": 4493 + "model_count": 4494 }, { "benchmark": "la_leaderboard", @@ -57,7 +57,7 @@ }, { "benchmark": "reward-bench", - "model_count": 327 + "model_count": 328 }, { "benchmark": "swe-bench", diff --git a/data/benchmarks/appworld_test_normal.json b/data/benchmarks/appworld_test_normal.json index df9e46daf5dd19a85e092a1af20da57930852dac..f3651490370442b9e2747ece9b07271eac5de3ad 100644 --- a/data/benchmarks/appworld_test_normal.json +++ b/data/benchmarks/appworld_test_normal.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "appworld/test_normal": 0.36 + "appworld/test_normal": 0.13 } }, { diff --git a/data/benchmarks/browsecompplus.json b/data/benchmarks/browsecompplus.json index 34f802eb19d92b7a54aef0531fdda5bcd88b81cf..130d44f42fc301b55e5c6bce906e1c1263721368 100644 --- a/data/benchmarks/browsecompplus.json +++ b/data/benchmarks/browsecompplus.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "browsecompplus": 0.61 + "browsecompplus": 0.49 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "browsecompplus": 0.26 + "browsecompplus": 0.48 } } ] diff --git a/data/benchmarks/hfopenllm_v2.json b/data/benchmarks/hfopenllm_v2.json index 69cfd9d7907951c5385529f14c35f4b9a0820d64..589e88495313e5113fd7851135b88baeeba06feb 100644 --- a/data/benchmarks/hfopenllm_v2.json +++ b/data/benchmarks/hfopenllm_v2.json @@ -1019,12 +1019,12 @@ "name": "Qwen2.5-1.5B-continuous-learnt", "developer": "AtAndDev", "scores": { - "IFEval": 0.4605, - "BBH": 0.4258, - "MATH Level 5": 0.0748, - "GPQA": 0.2659, - "MUSR": 0.3636, - "MMLU-PRO": 0.2812 + "IFEval": 0.4511, + "BBH": 0.4275, + "MATH Level 5": 0.1473, + "GPQA": 0.2701, + "MUSR": 0.3623, + "MMLU-PRO": 0.2806 } }, { @@ -1747,12 +1747,12 @@ "name": "NeuralDaredevil-SuperNova-Lite-7B-DARETIES-abliterated", "developer": "BoltMonkey", "scores": { - "IFEval": 0.7999, - "BBH": 0.5152, - "MATH Level 5": 0.1193, - "GPQA": 0.281, - "MUSR": 0.4019, - "MMLU-PRO": 0.3733 + "IFEval": 0.459, + "BBH": 0.5185, + "MATH Level 5": 0.0937, + "GPQA": 0.2743, + "MUSR": 0.4083, + "MMLU-PRO": 0.3631 } }, { @@ -3229,12 +3229,12 @@ "name": "PathfinderAI", "developer": "Daemontatox", "scores": { - "IFEval": 0.3745, - "BBH": 0.6668, - "MATH Level 5": 0.4758, - "GPQA": 0.3943, - "MUSR": 0.4858, - "MMLU-PRO": 0.5593 + "IFEval": 0.4855, + "BBH": 0.6627, + "MATH Level 5": 0.4841, + "GPQA": 0.3096, + "MUSR": 0.4256, + "MMLU-PRO": 0.5542 } }, { @@ -4009,12 +4009,12 @@ "name": "Llama-3.2-1B-SPIN-iter0", "developer": "DavieLion", "scores": { - "IFEval": 0.1549, - "BBH": 0.2937, - "MATH Level 5": 0.006, - "GPQA": 0.2576, + "IFEval": 0.1507, + "BBH": 0.293, + "MATH Level 5": 0.0, + "GPQA": 0.2534, "MUSR": 0.3565, - "MMLU-PRO": 0.1128 + "MMLU-PRO": 0.1125 } }, { @@ -4646,12 +4646,12 @@ "name": "MN-12B-LilithFrame", "developer": "DoppelReflEx", "scores": { - "IFEval": 0.436, - "BBH": 0.4956, - "MATH Level 5": 0.0589, - "GPQA": 0.3205, - "MUSR": 0.3843, - "MMLU-PRO": 0.3237 + "IFEval": 0.451, + "BBH": 0.4944, + "MATH Level 5": 0.1156, + "GPQA": 0.3196, + "MUSR": 0.3896, + "MMLU-PRO": 0.3256 } }, { @@ -8728,12 +8728,12 @@ "name": "Gemma-Ko-Merge-PEFT", "developer": "Gunulhona", "scores": { - "IFEval": 0.288, - "BBH": 0.5154, + "IFEval": 0.4441, + "BBH": 0.4863, "MATH Level 5": 0.0, - "GPQA": 0.3247, - "MUSR": 0.408, - "MMLU-PRO": 0.3817 + "GPQA": 0.307, + "MUSR": 0.3986, + "MMLU-PRO": 0.3098 } }, { @@ -9144,12 +9144,12 @@ "name": "SmolLM2-135M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.2883, - "BBH": 0.3124, - "MATH Level 5": 0.003, - "GPQA": 0.2357, - "MUSR": 0.3662, - "MMLU-PRO": 0.1115 + "IFEval": 0.0593, + "BBH": 0.3135, + "MATH Level 5": 0.0144, + "GPQA": 0.2341, + "MUSR": 0.3871, + "MMLU-PRO": 0.1092 } }, { @@ -9378,12 +9378,12 @@ "name": "JOSIEv4o-8b-stage1-v4", "developer": "Isaak-Carter", "scores": { - "IFEval": 0.2553, - "BBH": 0.4725, - "MATH Level 5": 0.0529, - "GPQA": 0.2919, - "MUSR": 0.3654, - "MMLU-PRO": 0.3316 + "IFEval": 0.2477, + "BBH": 0.4758, + "MATH Level 5": 0.0453, + "GPQA": 0.2911, + "MUSR": 0.3641, + "MMLU-PRO": 0.3292 } }, { @@ -13031,12 +13031,12 @@ "name": "SpydazWeb_AI_HumanAI_012_INSTRUCT_IA", "developer": "LeroyDyer", "scores": { - "IFEval": 0.3066, - "BBH": 0.4577, + "IFEval": 0.3036, + "BBH": 0.4575, "MATH Level 5": 0.0446, - "GPQA": 0.2995, - "MUSR": 0.4254, - "MMLU-PRO": 0.2318 + "GPQA": 0.3012, + "MUSR": 0.4253, + "MMLU-PRO": 0.2329 } }, { @@ -18141,11 +18141,11 @@ "developer": "PrimeIntellect", "scores": { "IFEval": 0.1757, - "BBH": 0.276, + "BBH": 0.274, "MATH Level 5": 0.0, - "GPQA": 0.2534, - "MUSR": 0.3339, - "MMLU-PRO": 0.1123 + "GPQA": 0.25, + "MUSR": 0.3753, + "MMLU-PRO": 0.112 } }, { @@ -18712,12 +18712,12 @@ "name": "ODB-14B-sce", "developer": "Quazim0t0", "scores": { - "IFEval": 0.7016, - "BBH": 0.6942, - "MATH Level 5": 0.4116, - "GPQA": 0.3624, - "MUSR": 0.4571, - "MMLU-PRO": 0.5411 + "IFEval": 0.2922, + "BBH": 0.6559, + "MATH Level 5": 0.2545, + "GPQA": 0.2659, + "MUSR": 0.3929, + "MMLU-PRO": 0.5207 } }, { @@ -19726,12 +19726,12 @@ "name": "Qwen2.5-Coder-7B-Instruct", "developer": "Qwen", "scores": { - "IFEval": 0.6101, - "BBH": 0.5008, - "MATH Level 5": 0.3716, - "GPQA": 0.2919, - "MUSR": 0.4073, - "MMLU-PRO": 0.3352 + "IFEval": 0.6147, + "BBH": 0.4999, + "MATH Level 5": 0.031, + "GPQA": 0.2936, + "MUSR": 0.4099, + "MMLU-PRO": 0.3354 } }, { @@ -25069,12 +25069,12 @@ "name": "Llama3.1-8B-Cobalt", "developer": "ValiantLabs", "scores": { - "IFEval": 0.3496, - "BBH": 0.4947, - "MATH Level 5": 0.1269, - "GPQA": 0.3037, - "MUSR": 0.3959, - "MMLU-PRO": 0.3644 + "IFEval": 0.7168, + "BBH": 0.4911, + "MATH Level 5": 0.1533, + "GPQA": 0.2861, + "MUSR": 0.3512, + "MMLU-PRO": 0.3663 } }, { @@ -25121,12 +25121,12 @@ "name": "Llama3.1-8B-ShiningValiant2", "developer": "ValiantLabs", "scores": { - "IFEval": 0.2678, - "BBH": 0.4429, - "MATH Level 5": 0.0521, - "GPQA": 0.302, - "MUSR": 0.3959, - "MMLU-PRO": 0.2927 + "IFEval": 0.6496, + "BBH": 0.4774, + "MATH Level 5": 0.0566, + "GPQA": 0.3104, + "MUSR": 0.3909, + "MMLU-PRO": 0.3382 } }, { @@ -26603,12 +26603,12 @@ "name": "QAIMath-Qwen2.5-7B-TIES", "developer": "adriszmar", "scores": { - "IFEval": 0.1685, - "BBH": 0.3124, - "MATH Level 5": 0.0015, - "GPQA": 0.2492, - "MUSR": 0.3963, - "MMLU-PRO": 0.1066 + "IFEval": 0.1746, + "BBH": 0.3126, + "MATH Level 5": 0.0, + "GPQA": 0.245, + "MUSR": 0.4096, + "MMLU-PRO": 0.1087 } }, { @@ -26954,12 +26954,12 @@ "name": "Llama-3.1-Tulu-3-8B", "developer": "allenai", "scores": { - "IFEval": 0.8255, - "BBH": 0.4061, - "MATH Level 5": 0.2115, - "GPQA": 0.297, + "IFEval": 0.8267, + "BBH": 0.405, + "MATH Level 5": 0.1964, + "GPQA": 0.2987, "MUSR": 0.4175, - "MMLU-PRO": 0.2821 + "MMLU-PRO": 0.2827 } }, { @@ -28449,11 +28449,11 @@ "name": "AMD-Llama-135m", "developer": "amd", "scores": { - "IFEval": 0.1842, - "BBH": 0.2974, - "MATH Level 5": 0.0053, - "GPQA": 0.2525, - "MUSR": 0.378, + "IFEval": 0.1918, + "BBH": 0.2969, + "MATH Level 5": 0.0076, + "GPQA": 0.2584, + "MUSR": 0.3846, "MMLU-PRO": 0.1169 } }, @@ -30360,12 +30360,12 @@ "name": "Llama-3.2-3B-Deep-Test", "developer": "bunnycore", "scores": { - "IFEval": 0.4652, - "BBH": 0.4531, - "MATH Level 5": 0.1284, - "GPQA": 0.2643, - "MUSR": 0.3394, - "MMLU-PRO": 0.3152 + "IFEval": 0.1775, + "BBH": 0.295, + "MATH Level 5": 0.0, + "GPQA": 0.2517, + "MUSR": 0.3647, + "MMLU-PRO": 0.1049 } }, { @@ -31790,12 +31790,12 @@ "name": "llama-43m-beta", "developer": "cpayne1303", "scores": { - "IFEval": 0.1949, - "BBH": 0.2965, - "MATH Level 5": 0.0045, + "IFEval": 0.1916, + "BBH": 0.2977, + "MATH Level 5": 0.0, "GPQA": 0.2685, - "MUSR": 0.3885, - "MMLU-PRO": 0.1111 + "MUSR": 0.3872, + "MMLU-PRO": 0.1132 } }, { @@ -32167,12 +32167,12 @@ "name": "Llama-3-8B-Orpo-v0.1", "developer": "dfurman", "scores": { - "IFEval": 0.2835, - "BBH": 0.3842, - "MATH Level 5": 0.0521, - "GPQA": 0.2609, - "MUSR": 0.3566, - "MMLU-PRO": 0.2298 + "IFEval": 0.3, + "BBH": 0.3853, + "MATH Level 5": 0.0415, + "GPQA": 0.2617, + "MUSR": 0.3579, + "MMLU-PRO": 0.2281 } }, { @@ -34572,12 +34572,12 @@ "name": "flan-t5-xl", "developer": "Google", "scores": { - "IFEval": 0.2207, - "BBH": 0.4537, - "MATH Level 5": 0.0008, - "GPQA": 0.2458, - "MUSR": 0.422, - "MMLU-PRO": 0.2142 + "IFEval": 0.2237, + "BBH": 0.4531, + "MATH Level 5": 0.0076, + "GPQA": 0.2525, + "MUSR": 0.4181, + "MMLU-PRO": 0.2147 } }, { @@ -34689,12 +34689,12 @@ "name": "gemma-2-2b-jpn-it", "developer": "Google", "scores": { - "IFEval": 0.5288, - "BBH": 0.4178, - "MATH Level 5": 0.0476, - "GPQA": 0.2752, - "MUSR": 0.3728, - "MMLU-PRO": 0.2467 + "IFEval": 0.5078, + "BBH": 0.4226, + "MATH Level 5": 0.0347, + "GPQA": 0.2852, + "MUSR": 0.3964, + "MMLU-PRO": 0.2578 } }, { @@ -37271,6 +37271,19 @@ "MMLU-PRO": 0.3103 } }, + { + "model_id": "icefog72/IceSakeV6RP-7b", + "name": "IceSakeV6RP-7b", + "developer": "icefog72", + "scores": { + "IFEval": 0.5033, + "BBH": 0.4976, + "MATH Level 5": 0.0619, + "GPQA": 0.2911, + "MUSR": 0.42, + "MMLU-PRO": 0.3093 + } + }, { "model_id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", @@ -42346,12 +42359,12 @@ "name": "Mistral-v0.3-7B-ORPO", "developer": "llmat", "scores": { - "IFEval": 0.377, - "BBH": 0.3978, - "MATH Level 5": 0.0242, - "GPQA": 0.2668, - "MUSR": 0.3555, - "MMLU-PRO": 0.2278 + "IFEval": 0.364, + "BBH": 0.4005, + "MATH Level 5": 0.0015, + "GPQA": 0.2693, + "MUSR": 0.3529, + "MMLU-PRO": 0.2301 } }, { @@ -43971,12 +43984,12 @@ "name": "Phi-3-mini-4k-instruct", "developer": "microsoft", "scores": { - "IFEval": 0.5613, - "BBH": 0.5676, - "MATH Level 5": 0.1163, - "GPQA": 0.3196, - "MUSR": 0.395, - "MMLU-PRO": 0.3866 + "IFEval": 0.5477, + "BBH": 0.5491, + "MATH Level 5": 0.1639, + "GPQA": 0.3322, + "MUSR": 0.4284, + "MMLU-PRO": 0.4022 } }, { @@ -44101,12 +44114,12 @@ "name": "phi-4", "developer": "microsoft", "scores": { - "IFEval": 0.0488, - "BBH": 0.6703, - "MATH Level 5": 0.2787, - "GPQA": 0.401, + "IFEval": 0.0585, + "BBH": 0.6691, + "MATH Level 5": 0.3165, + "GPQA": 0.406, "MUSR": 0.5034, - "MMLU-PRO": 0.5295 + "MMLU-PRO": 0.5287 } }, { @@ -44465,12 +44478,12 @@ "name": "Mixtral-8x7B-v0.1", "developer": "mistralai", "scores": { - "IFEval": 0.2326, - "BBH": 0.5098, - "MATH Level 5": 0.0937, - "GPQA": 0.3205, - "MUSR": 0.4413, - "MMLU-PRO": 0.3871 + "IFEval": 0.2415, + "BBH": 0.5087, + "MATH Level 5": 0.102, + "GPQA": 0.3138, + "MUSR": 0.4321, + "MMLU-PRO": 0.385 } }, { @@ -44725,12 +44738,12 @@ "name": "NeuralDaredevil-8B-abliterated", "developer": "mlabonne", "scores": { - "IFEval": 0.4162, - "BBH": 0.5124, - "MATH Level 5": 0.0853, - "GPQA": 0.3029, - "MUSR": 0.415, - "MMLU-PRO": 0.3802 + "IFEval": 0.7561, + "BBH": 0.5111, + "MATH Level 5": 0.0906, + "GPQA": 0.3062, + "MUSR": 0.4019, + "MMLU-PRO": 0.3841 } }, { @@ -45063,12 +45076,12 @@ "name": "Mistral-Nemo-Kurdish-Instruct", "developer": "nazimali", "scores": { - "IFEval": 0.486, - "BBH": 0.4721, - "MATH Level 5": 0.0846, - "GPQA": 0.2844, - "MUSR": 0.4006, - "MMLU-PRO": 0.3087 + "IFEval": 0.4964, + "BBH": 0.4699, + "MATH Level 5": 0.0045, + "GPQA": 0.2827, + "MUSR": 0.3979, + "MMLU-PRO": 0.3063 } }, { @@ -47598,12 +47611,12 @@ "name": "wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue", "developer": "ontocord", "scores": { - "IFEval": 0.1128, - "BBH": 0.3171, - "MATH Level 5": 0.0113, - "GPQA": 0.2685, - "MUSR": 0.346, - "MMLU-PRO": 0.1129 + "IFEval": 0.1162, + "BBH": 0.3184, + "MATH Level 5": 0.0076, + "GPQA": 0.2634, + "MUSR": 0.3447, + "MMLU-PRO": 0.1124 } }, { @@ -49431,12 +49444,12 @@ "name": "Calcium-Opus-14B-Elite", "developer": "prithivMLmods", "scores": { - "IFEval": 0.6064, - "BBH": 0.6296, - "MATH Level 5": 0.3708, - "GPQA": 0.3733, - "MUSR": 0.4873, - "MMLU-PRO": 0.5307 + "IFEval": 0.6052, + "BBH": 0.6317, + "MATH Level 5": 0.4789, + "GPQA": 0.3742, + "MUSR": 0.486, + "MMLU-PRO": 0.5302 } }, { @@ -50848,12 +50861,12 @@ "name": "Oracle-14B", "developer": "qingy2019", "scores": { - "IFEval": 0.2401, - "BBH": 0.4622, - "MATH Level 5": 0.0725, - "GPQA": 0.2609, - "MUSR": 0.3703, - "MMLU-PRO": 0.2379 + "IFEval": 0.2358, + "BBH": 0.4612, + "MATH Level 5": 0.0642, + "GPQA": 0.2576, + "MUSR": 0.3717, + "MMLU-PRO": 0.2382 } }, { @@ -50861,12 +50874,12 @@ "name": "Qwen2.5-Math-14B-Instruct", "developer": "qingy2019", "scores": { - "IFEval": 0.6066, - "BBH": 0.635, - "MATH Level 5": 0.3716, - "GPQA": 0.3725, + "IFEval": 0.6005, + "BBH": 0.6356, + "MATH Level 5": 0.2764, + "GPQA": 0.3691, "MUSR": 0.4757, - "MMLU-PRO": 0.5331 + "MMLU-PRO": 0.5339 } }, { @@ -51589,12 +51602,12 @@ "name": "Rombos-LLM-V2.5.1-Qwen-3b", "developer": "rombodawg", "scores": { - "IFEval": 0.2595, - "BBH": 0.3884, - "MATH Level 5": 0.0914, - "GPQA": 0.2743, + "IFEval": 0.2566, + "BBH": 0.39, + "MATH Level 5": 0.1208, + "GPQA": 0.2626, "MUSR": 0.3991, - "MMLU-PRO": 0.2719 + "MMLU-PRO": 0.2741 } }, { @@ -54332,12 +54345,12 @@ "name": "lambda-gemma-2-9b-dpo", "developer": "tanliboy", "scores": { - "IFEval": 0.1829, - "BBH": 0.5488, - "MATH Level 5": 0.0, - "GPQA": 0.3104, - "MUSR": 0.4056, - "MMLU-PRO": 0.3805 + "IFEval": 0.4501, + "BBH": 0.5472, + "MATH Level 5": 0.0944, + "GPQA": 0.3138, + "MUSR": 0.4017, + "MMLU-PRO": 0.3792 } }, { @@ -56932,12 +56945,12 @@ "name": "Hebrew-Mistral-7B-200K", "developer": "yam-peleg", "scores": { - "IFEval": 0.177, - "BBH": 0.3411, - "MATH Level 5": 0.031, - "GPQA": 0.2534, - "MUSR": 0.374, - "MMLU-PRO": 0.2529 + "IFEval": 0.1856, + "BBH": 0.4149, + "MATH Level 5": 0.0234, + "GPQA": 0.276, + "MUSR": 0.3765, + "MMLU-PRO": 0.2573 } }, { @@ -56984,12 +56997,12 @@ "name": "BagelMIsteryTour-v2-8x7B", "developer": "ycros", "scores": { - "IFEval": 0.5994, - "BBH": 0.5159, - "MATH Level 5": 0.0785, - "GPQA": 0.3045, - "MUSR": 0.4203, - "MMLU-PRO": 0.3473 + "IFEval": 0.6262, + "BBH": 0.5142, + "MATH Level 5": 0.0937, + "GPQA": 0.3079, + "MUSR": 0.4138, + "MMLU-PRO": 0.3481 } }, { diff --git a/data/benchmarks/livecodebenchpro.json b/data/benchmarks/livecodebenchpro.json index f8f77204727bc54bca1bbaaa204ca44d2b265f12..449e36cb0d36e98776192be2ede4fd274d74ec7e 100644 --- a/data/benchmarks/livecodebenchpro.json +++ b/data/benchmarks/livecodebenchpro.json @@ -255,9 +255,9 @@ "name": "o4-mini-2025-04-16", "developer": "OpenAI", "scores": { - "Hard Problems": 0.0143, - "Medium Problems": 0.2923, - "Easy Problems": 0.8571 + "Hard Problems": 0.014084507042253521, + "Medium Problems": 0.30985915492957744, + "Easy Problems": 0.8873239436619719 } }, { diff --git a/data/benchmarks/reward-bench.json b/data/benchmarks/reward-bench.json index 71717215bb28d07da86d76055db6b05b5dbcd648..435eb05406175ffe8530e3141464bcd0e3abcc31 100644 --- a/data/benchmarks/reward-bench.json +++ b/data/benchmarks/reward-bench.json @@ -81,17 +81,17 @@ "name": "CIR-AMS/BTRM_Qwen2_7b_0613", "developer": "CIR-AMS", "scores": { - "Score": 0.5736, - "Chat": 0.9749, - "Chat Hard": 0.5724, - "Safety": 0.7178, - "Reasoning": 0.8775, - "Prior Sets (0.5 weight)": 0.7029, + "Score": 0.8172, "Factuality": 0.5347, "Precise IF": 0.3563, "Math": 0.6066, + "Safety": 0.9014, "Focus": 0.5737, - "Ties": 0.6527 + "Ties": 0.6527, + "Chat": 0.9749, + "Chat Hard": 0.5724, + "Reasoning": 0.8775, + "Prior Sets (0.5 weight)": 0.7029 } }, { @@ -499,17 +499,17 @@ "name": "Nexusflow/Starling-RM-34B", "developer": "Nexusflow", "scores": { - "Score": 0.8133, + "Score": 0.4553, + "Chat": 0.9693, + "Chat Hard": 0.5724, + "Safety": 0.7556, + "Reasoning": 0.8845, + "Prior Sets (0.5 weight)": 0.7137, "Factuality": 0.4589, "Precise IF": 0.3187, "Math": 0.6175, - "Safety": 0.877, "Focus": 0.4808, - "Ties": 0.1004, - "Chat": 0.9693, - "Chat Hard": 0.5724, - "Reasoning": 0.8845, - "Prior Sets (0.5 weight)": 0.7137 + "Ties": 0.1004 } }, { @@ -555,17 +555,17 @@ "name": "OpenAssistant/oasst-rm-2-pythia-6.9b-epoch-1", "developer": "OpenAssistant", "scores": { - "Score": 0.615, + "Score": 0.2653, + "Chat": 0.9246, + "Chat Hard": 0.3728, + "Safety": 0.3289, + "Reasoning": 0.5855, + "Prior Sets (0.5 weight)": 0.6801, "Factuality": 0.3979, "Precise IF": 0.2875, "Math": 0.377, - "Safety": 0.5446, "Focus": 0.1535, - "Ties": 0.047, - "Chat": 0.9246, - "Chat Hard": 0.3728, - "Reasoning": 0.5855, - "Prior Sets (0.5 weight)": 0.6801 + "Ties": 0.047 } }, { @@ -573,17 +573,17 @@ "name": "OpenAssistant/oasst-rm-2.1-pythia-1.4b-epoch-2.5", "developer": "OpenAssistant", "scores": { - "Score": 0.6901, + "Score": 0.2648, + "Chat": 0.8855, + "Chat Hard": 0.4868, + "Safety": 0.3244, + "Reasoning": 0.7752, + "Prior Sets (0.5 weight)": 0.6533, "Factuality": 0.3179, "Precise IF": 0.2625, "Math": 0.3934, - "Safety": 0.6311, "Focus": 0.2707, - "Ties": 0.0198, - "Chat": 0.8855, - "Chat Hard": 0.4868, - "Reasoning": 0.7752, - "Prior Sets (0.5 weight)": 0.6533 + "Ties": 0.0198 } }, { @@ -591,17 +591,17 @@ "name": "OpenAssistant/reward-model-deberta-v3-large-v2", "developer": "OpenAssistant", "scores": { - "Score": 0.6126, + "Score": 0.32, + "Chat": 0.8939, + "Chat Hard": 0.4518, + "Safety": 0.3667, + "Reasoning": 0.3855, + "Prior Sets (0.5 weight)": 0.5836, "Factuality": 0.3853, "Precise IF": 0.2687, "Math": 0.5027, - "Safety": 0.7338, "Focus": 0.2768, - "Ties": 0.12, - "Chat": 0.8939, - "Chat Hard": 0.4518, - "Reasoning": 0.3855, - "Prior Sets (0.5 weight)": 0.5836 + "Ties": 0.12 } }, { @@ -627,17 +627,17 @@ "name": "PKU-Alignment/beaver-7b-v1.0-reward", "developer": "PKU-Alignment", "scores": { - "Score": 0.4727, + "Score": 0.1606, + "Chat": 0.8184, + "Chat Hard": 0.2873, + "Safety": 0.1422, + "Reasoning": 0.346, + "Prior Sets (0.5 weight)": 0.5993, "Factuality": 0.2105, "Precise IF": 0.2938, "Math": 0.2623, - "Safety": 0.3757, "Focus": 0.0646, - "Ties": -0.01, - "Chat": 0.8184, - "Chat Hard": 0.2873, - "Reasoning": 0.346, - "Prior Sets (0.5 weight)": 0.5993 + "Ties": -0.01 } }, { @@ -938,17 +938,17 @@ "name": "Ray2333/GRM-llama3-8B-distill", "developer": "Ray2333", "scores": { - "Score": 0.589, - "Chat": 0.9832, - "Chat Hard": 0.6842, - "Safety": 0.7222, - "Reasoning": 0.9133, - "Prior Sets (0.5 weight)": 0.7209, + "Score": 0.8464, "Factuality": 0.5874, "Precise IF": 0.3875, "Math": 0.5902, + "Safety": 0.8676, "Focus": 0.6727, - "Ties": 0.5743 + "Ties": 0.5743, + "Chat": 0.9832, + "Chat Hard": 0.6842, + "Reasoning": 0.9133, + "Prior Sets (0.5 weight)": 0.7209 } }, { @@ -956,17 +956,17 @@ "name": "Ray2333/GRM-llama3-8B-sftreg", "developer": "Ray2333", "scores": { - "Score": 0.6089, - "Chat": 0.986, - "Chat Hard": 0.6776, - "Safety": 0.7867, - "Reasoning": 0.9229, - "Prior Sets (0.5 weight)": 0.7309, + "Score": 0.8542, "Factuality": 0.6189, "Precise IF": 0.3875, "Math": 0.5792, + "Safety": 0.8919, "Focus": 0.6828, - "Ties": 0.5981 + "Ties": 0.5981, + "Chat": 0.986, + "Chat Hard": 0.6776, + "Reasoning": 0.9229, + "Prior Sets (0.5 weight)": 0.7309 } }, { @@ -1098,16 +1098,16 @@ "name": "ShikaiChen/LDL-Reward-Gemma-2-27B-v0.1", "developer": "ShikaiChen", "scores": { - "Score": 0.7249, - "Chat": 0.9637, - "Chat Hard": 0.9079, - "Safety": 0.9222, - "Reasoning": 0.9903, + "Score": 0.9499, "Factuality": 0.7558, "Precise IF": 0.35, "Math": 0.6448, + "Safety": 0.9378, "Focus": 0.9131, - "Ties": 0.7633 + "Ties": 0.7633, + "Chat": 0.9637, + "Chat Hard": 0.9079, + "Reasoning": 0.9903 } }, { @@ -1156,16 +1156,16 @@ "name": "Skywork/Skywork-Reward-Gemma-2-27B-v0.2", "developer": "Skywork", "scores": { - "Score": 0.9426, + "Score": 0.7531, + "Chat": 0.9609, + "Chat Hard": 0.8991, + "Safety": 0.9689, + "Reasoning": 0.9807, "Factuality": 0.7674, "Precise IF": 0.375, "Math": 0.6721, - "Safety": 0.9297, "Focus": 0.9172, - "Ties": 0.8182, - "Chat": 0.9609, - "Chat Hard": 0.8991, - "Reasoning": 0.9807 + "Ties": 0.8182 } }, { @@ -1173,16 +1173,16 @@ "name": "Skywork/Skywork-Reward-Llama-3.1-8B", "developer": "Skywork", "scores": { - "Score": 0.9252, + "Score": 0.7314, + "Chat": 0.9581, + "Chat Hard": 0.8728, + "Safety": 0.9333, + "Reasoning": 0.962, "Factuality": 0.6989, "Precise IF": 0.425, "Math": 0.6284, - "Safety": 0.9081, "Focus": 0.9616, - "Ties": 0.741, - "Chat": 0.9581, - "Chat Hard": 0.8728, - "Reasoning": 0.962 + "Ties": 0.741 } }, { @@ -1379,10 +1379,10 @@ "name": "ai2/tulu-2-7b-rm-v0-nectar-binarized-3.8m-check...", "developer": "AI2", "scores": { - "Score": 0.6895, + "Score": 0.7008, "Chat": 0.9385, - "Chat Hard": 0.3706, - "Safety": 0.7595 + "Chat Hard": 0.3882, + "Safety": 0.7757 } }, { @@ -1423,17 +1423,17 @@ "name": "allenai/Llama-3.1-70B-Instruct-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.7606, - "Chat": 0.9665, - "Chat Hard": 0.8355, - "Safety": 0.8844, - "Reasoning": 0.8969, - "Prior Sets (0.5 weight)": 0.0, + "Score": 0.9021, "Factuality": 0.8126, "Precise IF": 0.4188, "Math": 0.6995, + "Safety": 0.9095, "Focus": 0.8646, - "Ties": 0.8835 + "Ties": 0.8835, + "Chat": 0.9665, + "Chat Hard": 0.8355, + "Reasoning": 0.8969, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1495,17 +1495,17 @@ "name": "allenai/Llama-3.1-Tulu-3-8B-DPO-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.687, - "Chat": 0.9553, - "Chat Hard": 0.761, - "Safety": 0.86, - "Reasoning": 0.7898, - "Prior Sets (0.5 weight)": 0.0, + "Score": 0.8431, "Factuality": 0.7516, "Precise IF": 0.3875, "Math": 0.6284, + "Safety": 0.8662, "Focus": 0.8545, - "Ties": 0.6397 + "Ties": 0.6397, + "Chat": 0.9553, + "Chat Hard": 0.761, + "Reasoning": 0.7898, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1545,17 +1545,17 @@ "name": "allenai/Llama-3.1-Tulu-3-8B-SFT-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8551, + "Score": 0.6821, + "Chat": 0.9497, + "Chat Hard": 0.7917, + "Safety": 0.8978, + "Reasoning": 0.8005, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.7326, "Precise IF": 0.3875, "Math": 0.5792, - "Safety": 0.8784, "Focus": 0.8889, - "Ties": 0.6063, - "Chat": 0.9497, - "Chat Hard": 0.7917, - "Reasoning": 0.8005, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.6063 } }, { @@ -2085,6 +2085,20 @@ "Ties": 0.3534 } }, + { + "model_id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "developer": "allenai", + "scores": { + "Score": 0.5151, + "Factuality": 0.6484, + "Precise IF": 0.3312, + "Math": 0.5574, + "Safety": 0.7289, + "Focus": 0.4889, + "Ties": 0.3357 + } + }, { "model_id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", @@ -3441,16 +3455,16 @@ "name": "Claude 3.5 Sonnet 20240620", "developer": "Anthropic", "scores": { - "Score": 0.6466, - "Chat": 0.9637, - "Chat Hard": 0.7401, - "Safety": 0.8519, - "Reasoning": 0.8469, + "Score": 0.8417, "Factuality": 0.5284, "Precise IF": 0.3875, "Math": 0.5683, + "Safety": 0.8162, "Focus": 0.8697, - "Ties": 0.674 + "Ties": 0.674, + "Chat": 0.9637, + "Chat Hard": 0.7401, + "Reasoning": 0.8469 } }, { @@ -3472,17 +3486,17 @@ "name": "Claude 3 Haiku 20240307", "developer": "Anthropic", "scores": { - "Score": 0.7289, + "Score": 0.3711, + "Chat": 0.9274, + "Chat Hard": 0.5197, + "Safety": 0.595, + "Reasoning": 0.706, + "Prior Sets (0.5 weight)": 0.6635, "Factuality": 0.4042, "Precise IF": 0.2812, "Math": 0.3552, - "Safety": 0.7953, "Focus": 0.501, - "Ties": 0.0899, - "Chat": 0.9274, - "Chat Hard": 0.5197, - "Reasoning": 0.706, - "Prior Sets (0.5 weight)": 0.6635 + "Ties": 0.0899 } }, { @@ -3490,16 +3504,16 @@ "name": "Claude 3 Opus 20240229", "developer": "Anthropic", "scores": { - "Score": 0.8008, + "Score": 0.5744, + "Chat": 0.9469, + "Chat Hard": 0.6031, + "Safety": 0.8378, + "Reasoning": 0.7868, "Factuality": 0.5389, "Precise IF": 0.3312, "Math": 0.5137, - "Safety": 0.8662, "Focus": 0.6646, - "Ties": 0.5601, - "Chat": 0.9469, - "Chat Hard": 0.6031, - "Reasoning": 0.7868 + "Ties": 0.5601 } }, { @@ -3752,17 +3766,17 @@ "name": "hendrydong/Mistral-RM-for-RAFT-GSHF-v0", "developer": "hendrydong", "scores": { - "Score": 0.5851, - "Chat": 0.9832, - "Chat Hard": 0.5789, - "Safety": 0.6956, - "Reasoning": 0.7434, - "Prior Sets (0.5 weight)": 0.7508, + "Score": 0.7847, "Factuality": 0.5779, "Precise IF": 0.3625, "Math": 0.6011, + "Safety": 0.85, "Focus": 0.6747, - "Ties": 0.5988 + "Ties": 0.5988, + "Chat": 0.9832, + "Chat Hard": 0.5789, + "Reasoning": 0.7434, + "Prior Sets (0.5 weight)": 0.7508 } }, { @@ -3787,16 +3801,16 @@ "name": "internlm/internlm2-1_8b-reward", "developer": "internlm", "scores": { - "Score": 0.8217, + "Score": 0.3902, + "Chat": 0.9358, + "Chat Hard": 0.6623, + "Safety": 0.4711, + "Reasoning": 0.8724, "Factuality": 0.2758, "Precise IF": 0.3625, "Math": 0.4426, - "Safety": 0.8162, "Focus": 0.596, - "Ties": 0.1934, - "Chat": 0.9358, - "Chat Hard": 0.6623, - "Reasoning": 0.8724 + "Ties": 0.1934 } }, { @@ -3804,16 +3818,16 @@ "name": "internlm/internlm2-20b-reward", "developer": "internlm", "scores": { - "Score": 0.5628, - "Chat": 0.9888, - "Chat Hard": 0.7654, - "Safety": 0.6111, - "Reasoning": 0.9576, + "Score": 0.9016, "Factuality": 0.5558, "Precise IF": 0.3625, "Math": 0.5738, + "Safety": 0.8946, "Focus": 0.7253, - "Ties": 0.5483 + "Ties": 0.5483, + "Chat": 0.9888, + "Chat Hard": 0.7654, + "Reasoning": 0.9576 } }, { @@ -3821,16 +3835,16 @@ "name": "internlm/internlm2-7b-reward", "developer": "internlm", "scores": { - "Score": 0.8759, + "Score": 0.5335, + "Chat": 0.9916, + "Chat Hard": 0.6952, + "Safety": 0.5956, + "Reasoning": 0.9453, "Factuality": 0.4211, "Precise IF": 0.4, "Math": 0.5628, - "Safety": 0.8716, "Focus": 0.7051, - "Ties": 0.5164, - "Chat": 0.9916, - "Chat Hard": 0.6952, - "Reasoning": 0.9453 + "Ties": 0.5164 } }, { @@ -4041,16 +4055,16 @@ "name": "nicolinho/QRM-Llama3.1-8B-v2", "developer": "nicolinho", "scores": { - "Score": 0.9314, + "Score": 0.7074, + "Chat": 0.9637, + "Chat Hard": 0.8684, + "Safety": 0.9467, + "Reasoning": 0.9677, "Factuality": 0.6653, "Precise IF": 0.4062, "Math": 0.612, - "Safety": 0.9257, "Focus": 0.8909, - "Ties": 0.7234, - "Chat": 0.9637, - "Chat Hard": 0.8684, - "Reasoning": 0.9677 + "Ties": 0.7234 } }, { @@ -4188,16 +4202,16 @@ "name": "GPT-4o 2024-08-06", "developer": "OpenAI", "scores": { - "Score": 0.6493, - "Chat": 0.9609, - "Chat Hard": 0.761, - "Safety": 0.8619, - "Reasoning": 0.8661, + "Score": 0.8673, "Factuality": 0.5684, "Precise IF": 0.3312, "Math": 0.623, + "Safety": 0.8811, "Focus": 0.7293, - "Ties": 0.7819 + "Ties": 0.7819, + "Chat": 0.9609, + "Chat Hard": 0.761, + "Reasoning": 0.8661 } }, { @@ -4205,16 +4219,16 @@ "name": "GPT-4o mini 2024-07-18", "developer": "OpenAI", "scores": { - "Score": 0.5796, - "Chat": 0.9497, - "Chat Hard": 0.6075, - "Safety": 0.7667, - "Reasoning": 0.8374, + "Score": 0.8007, "Factuality": 0.4105, "Precise IF": 0.3438, "Math": 0.5191, + "Safety": 0.8081, "Focus": 0.7414, - "Ties": 0.6962 + "Ties": 0.6962, + "Chat": 0.9497, + "Chat Hard": 0.6075, + "Reasoning": 0.8374 } }, { @@ -4235,17 +4249,17 @@ "name": "openbmb/Eurus-RM-7b", "developer": "openbmb", "scores": { - "Score": 0.8159, + "Score": 0.5806, + "Chat": 0.9804, + "Chat Hard": 0.6557, + "Safety": 0.6267, + "Reasoning": 0.8633, + "Prior Sets (0.5 weight)": 0.7172, "Factuality": 0.6, "Precise IF": 0.3438, "Math": 0.5683, - "Safety": 0.8135, "Focus": 0.7475, - "Ties": 0.5972, - "Chat": 0.9804, - "Chat Hard": 0.6557, - "Reasoning": 0.8633, - "Prior Sets (0.5 weight)": 0.7172 + "Ties": 0.5972 } }, { @@ -4266,17 +4280,17 @@ "name": "openbmb/UltraRM-13b", "developer": "openbmb", "scores": { - "Score": 0.6903, + "Score": 0.4683, + "Chat": 0.9637, + "Chat Hard": 0.5548, + "Safety": 0.5089, + "Reasoning": 0.6244, + "Prior Sets (0.5 weight)": 0.7294, "Factuality": 0.5063, "Precise IF": 0.3312, "Math": 0.5519, - "Safety": 0.5986, "Focus": 0.6081, - "Ties": 0.3036, - "Chat": 0.9637, - "Chat Hard": 0.5548, - "Reasoning": 0.6244, - "Prior Sets (0.5 weight)": 0.7294 + "Ties": 0.3036 } }, { @@ -4356,17 +4370,17 @@ "name": "sfairXC/FsfairX-LLaMA3-RM-v0.1", "developer": "sfairXC", "scores": { - "Score": 0.6292, - "Chat": 0.9944, - "Chat Hard": 0.6513, - "Safety": 0.7667, - "Reasoning": 0.8644, - "Prior Sets (0.5 weight)": 0.7492, + "Score": 0.8338, "Factuality": 0.5916, "Precise IF": 0.4188, "Math": 0.6284, + "Safety": 0.8676, "Focus": 0.7051, - "Ties": 0.6647 + "Ties": 0.6647, + "Chat": 0.9944, + "Chat Hard": 0.6513, + "Reasoning": 0.8644, + "Prior Sets (0.5 weight)": 0.7492 } }, { diff --git a/data/benchmarks/swe-bench.json b/data/benchmarks/swe-bench.json index 88b6df39554e558930cdff771fd1ab7d2dd73e3a..093176be195c26e24230b48d3e9b488139a6d6bf 100644 --- a/data/benchmarks/swe-bench.json +++ b/data/benchmarks/swe-bench.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "swe-bench": 0.6061 + "swe-bench": 0.8072 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "swe-bench": 0.71 + "swe-bench": 0.7234 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "swe-bench": 0.57 + "swe-bench": 0.5455 } } ] diff --git a/data/benchmarks/tau-bench-2_airline.json b/data/benchmarks/tau-bench-2_airline.json index 3829696a07bc037642c69851a891e0aeb0e5febf..203ba536a38c5fd6f5a3b4f5a7c625bb98c214a5 100644 --- a/data/benchmarks/tau-bench-2_airline.json +++ b/data/benchmarks/tau-bench-2_airline.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/airline": 0.72 + "tau-bench-2/airline": 0.66 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/airline": 0.68 + "tau-bench-2/airline": 0.7 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/airline": 0.6 + "tau-bench-2/airline": 0.48 } } ] diff --git a/data/benchmarks/tau-bench-2_retail.json b/data/benchmarks/tau-bench-2_retail.json index 220f36fbae7a50d2d3974e0cd62ceacad728d0fd..bb3c509d520d0e5ab84b4c3e44db8c84041b45fa 100644 --- a/data/benchmarks/tau-bench-2_retail.json +++ b/data/benchmarks/tau-bench-2_retail.json @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/retail": 0.68 + "tau-bench-2/retail": 0.51 } } ] diff --git a/data/benchmarks/tau-bench-2_telecom.json b/data/benchmarks/tau-bench-2_telecom.json index 5e2e97c5a63c814404bfd0e936bb7f41ce63593e..0371242e1ef28be51ed37461c45171b5f6b938db 100644 --- a/data/benchmarks/tau-bench-2_telecom.json +++ b/data/benchmarks/tau-bench-2_telecom.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/telecom": 0.76 + "tau-bench-2/telecom": 0.84 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/telecom": 0.73 + "tau-bench-2/telecom": 0.6852 } }, { diff --git a/data/benchmarks/terminal-bench-2.0.json b/data/benchmarks/terminal-bench-2.0.json index eedfc30dd08733e57be031879da03991ca05bc76..a8e801f280cc4af65aa221436665986581643fe9 100644 --- a/data/benchmarks/terminal-bench-2.0.json +++ b/data/benchmarks/terminal-bench-2.0.json @@ -5,7 +5,7 @@ "name": "Qwen 3 Coder 480B", "developer": "Alibaba", "scores": { - "terminal-bench-2.0": 25.4 + "terminal-bench-2.0": 23.9 } }, { @@ -13,7 +13,7 @@ "name": "Claude Haiku 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 27.5 + "terminal-bench-2.0": 35.5 } }, { @@ -21,7 +21,7 @@ "name": "Claude Opus 4.1", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 34.8 + "terminal-bench-2.0": 38.0 } }, { @@ -29,7 +29,7 @@ "name": "Claude Opus 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 52.1 + "terminal-bench-2.0": 59.1 } }, { @@ -37,7 +37,7 @@ "name": "Claude Opus 4.6", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 74.7 + "terminal-bench-2.0": 66.9 } }, { @@ -45,7 +45,7 @@ "name": "Claude Sonnet 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 42.5 + "terminal-bench-2.0": 40.1 } }, { @@ -61,7 +61,7 @@ "name": "Gemini 2.5 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 16.9 + "terminal-bench-2.0": 17.1 } }, { @@ -69,7 +69,7 @@ "name": "Gemini 2.5 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 16.4 + "terminal-bench-2.0": 19.6 } }, { @@ -85,7 +85,7 @@ "name": "Gemini 3 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 56.9 + "terminal-bench-2.0": 62.2 } }, { @@ -149,7 +149,7 @@ "name": "Multiple", "developer": "Multiple", "scores": { - "terminal-bench-2.0": 71.0 + "terminal-bench-2.0": 58.4 } }, { @@ -165,7 +165,7 @@ "name": "GPT-5-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 43.4 + "terminal-bench-2.0": 41.3 } }, { @@ -173,7 +173,7 @@ "name": "GPT-5-Mini", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 31.9 + "terminal-bench-2.0": 22.2 } }, { @@ -181,7 +181,7 @@ "name": "GPT-5-Nano", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 7.0 + "terminal-bench-2.0": 7.9 } }, { @@ -197,7 +197,7 @@ "name": "GPT-5.1-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 53.5 + "terminal-bench-2.0": 36.9 } }, { @@ -221,7 +221,7 @@ "name": "GPT-5.2", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 60.7 + "terminal-bench-2.0": 54.0 } }, { @@ -237,7 +237,7 @@ "name": "GPT-5.3-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 74.6 + "terminal-bench-2.0": 77.3 } }, { @@ -261,7 +261,7 @@ "name": "Grok 4", "developer": "xAI", "scores": { - "terminal-bench-2.0": 23.1 + "terminal-bench-2.0": 25.4 } }, { @@ -285,7 +285,7 @@ "name": "GLM 4.7", "developer": "Z-AI", "scores": { - "terminal-bench-2.0": 33.4 + "terminal-bench-2.0": 33.3 } }, { diff --git a/data/developers.json b/data/developers.json index 013b8e244887bf6c63a6c680ce52cd50fedc7ed7..003e462b3fb10e7ebce2d70b3ed9418882ec3a93 100644 --- a/data/developers.json +++ b/data/developers.json @@ -173,7 +173,7 @@ }, { "developer": "allenai", - "model_count": 161 + "model_count": 162 }, { "developer": "allknowingroger", @@ -1097,7 +1097,7 @@ }, { "developer": "icefog72", - "model_count": 61 + "model_count": 62 }, { "developer": "IDEA-CCNL", diff --git a/data/developers/adriszmar.json b/data/developers/adriszmar.json index acb90d745752909d8f96f323acb9caa9b19061ae..1f1d39916960942963a9c3c265196aea3657be38 100644 --- a/data/developers/adriszmar.json +++ b/data/developers/adriszmar.json @@ -7,12 +7,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1685, - "hfopenllm_v2/BBH": 0.3124, - "hfopenllm_v2/MATH Level 5": 0.0015, - "hfopenllm_v2/GPQA": 0.2492, - "hfopenllm_v2/MUSR": 0.3963, - "hfopenllm_v2/MMLU-PRO": 0.1066 + "hfopenllm_v2/IFEval": 0.1746, + "hfopenllm_v2/BBH": 0.3126, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.245, + "hfopenllm_v2/MUSR": 0.4096, + "hfopenllm_v2/MMLU-PRO": 0.1087 } } ] diff --git a/data/developers/ai2.json b/data/developers/ai2.json index 498b6154facf655073d48826dd02116d29e34e45..4934c11b7806e647da8c3821dcfccba2566bb947 100644 --- a/data/developers/ai2.json +++ b/data/developers/ai2.json @@ -43,10 +43,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6895, + "reward-bench/Score": 0.7008, "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3706, - "reward-bench/Safety": 0.7595 + "reward-bench/Chat Hard": 0.3882, + "reward-bench/Safety": 0.7757 } }, { diff --git a/data/developers/alibaba.json b/data/developers/alibaba.json index 8efe41024b181a324827f30ff358f0234de7b987..759f3a2b29b65da87e85bd7df67845381cdc3145 100644 --- a/data/developers/alibaba.json +++ b/data/developers/alibaba.json @@ -7,7 +7,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 25.4 + "terminal-bench-2.0/terminal-bench-2.0": 23.9 } }, { diff --git a/data/developers/allenai.json b/data/developers/allenai.json index 1c85dbe2cfcc674683235ce774bcd2bc78253288..26883d607b7a5071bd9da08389772de32edfe871 100644 --- a/data/developers/allenai.json +++ b/data/developers/allenai.json @@ -63,17 +63,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7606, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Safety": 0.8844, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.9021, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835 + "reward-bench/Ties": 0.8835, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -181,12 +181,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { @@ -209,17 +209,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.687, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Safety": 0.86, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8431, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397 + "reward-bench/Ties": 0.6397, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -282,17 +282,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8551, + "reward-bench/Score": 0.6821, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.7917, + "reward-bench/Safety": 0.8978, + "reward-bench/Reasoning": 0.8005, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7326, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8784, "reward-bench/Focus": 0.8889, - "reward-bench/Ties": 0.6063, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.7917, - "reward-bench/Reasoning": 0.8005, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6063 } }, { @@ -1054,6 +1054,21 @@ "reward-bench/Ties": 0.3534 } }, + { + "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "developer": "allenai", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.5151, + "reward-bench/Factuality": 0.6484, + "reward-bench/Precise IF": 0.3312, + "reward-bench/Math": 0.5574, + "reward-bench/Safety": 0.7289, + "reward-bench/Focus": 0.4889, + "reward-bench/Ties": 0.3357 + } + }, { "id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", diff --git a/data/developers/amd.json b/data/developers/amd.json index ec85a4364a761e30f1999f359d0d247d8857e139..f58c5a6931b8f7bef3756ba18944919a5d99e792 100644 --- a/data/developers/amd.json +++ b/data/developers/amd.json @@ -7,11 +7,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1842, - "hfopenllm_v2/BBH": 0.2974, - "hfopenllm_v2/MATH Level 5": 0.0053, - "hfopenllm_v2/GPQA": 0.2525, - "hfopenllm_v2/MUSR": 0.378, + "hfopenllm_v2/IFEval": 0.1918, + "hfopenllm_v2/BBH": 0.2969, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/MUSR": 0.3846, "hfopenllm_v2/MMLU-PRO": 0.1169 } } diff --git a/data/developers/anthropic.json b/data/developers/anthropic.json index 819a852f845410b5f7876451b1dcb93ba59f6701..c1b51c55828f681c4d430c9fc7c3821e0c017451 100644 --- a/data/developers/anthropic.json +++ b/data/developers/anthropic.json @@ -204,16 +204,16 @@ "helm_mmlu/Virology": 0.602, "helm_mmlu/World Religions": 0.924, "helm_mmlu/Mean win rate": 0.17, - "reward-bench/Score": 0.6466, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.7401, - "reward-bench/Safety": 0.8519, - "reward-bench/Reasoning": 0.8469, + "reward-bench/Score": 0.8417, "reward-bench/Factuality": 0.5284, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.8697, - "reward-bench/Ties": 0.674 + "reward-bench/Ties": 0.674, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.7401, + "reward-bench/Reasoning": 0.8469 } }, { @@ -371,17 +371,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.7289, + "reward-bench/Score": 0.3711, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Safety": 0.595, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, - "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635 + "reward-bench/Ties": 0.0899 } }, { @@ -436,16 +436,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.8008, + "reward-bench/Score": 0.5744, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.8378, + "reward-bench/Reasoning": 0.7868, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Reasoning": 0.7868 + "reward-bench/Ties": 0.5601 } }, { @@ -525,7 +525,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.5 + "terminal-bench-2.0/terminal-bench-2.0": 35.5 } }, { @@ -650,12 +650,12 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { + "browsecompplus/browsecompplus": 0.49, "appworld_test_normal/appworld/test_normal": 0.64, - "browsecompplus/browsecompplus": 0.61, - "swe-bench/swe-bench": 0.6061, - "tau-bench-2_airline/tau-bench-2/airline": 0.72, + "swe-bench/swe-bench": 0.8072, + "tau-bench-2_airline/tau-bench-2/airline": 0.66, "tau-bench-2_retail/tau-bench-2/retail": 0.85, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } }, { @@ -664,7 +664,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 34.8 + "terminal-bench-2.0/terminal-bench-2.0": 38.0 } }, { @@ -673,7 +673,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 52.1 + "terminal-bench-2.0/terminal-bench-2.0": 59.1 } }, { @@ -682,7 +682,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.7 + "terminal-bench-2.0/terminal-bench-2.0": 66.9 } }, { @@ -756,7 +756,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.5 + "terminal-bench-2.0/terminal-bench-2.0": 40.1 } }, { diff --git a/data/developers/atanddev.json b/data/developers/atanddev.json index 42530fda0188ea30162bbff58f1010442b0394c9..d269c1fbba37e9ea4df30635a880656764997806 100644 --- a/data/developers/atanddev.json +++ b/data/developers/atanddev.json @@ -7,12 +7,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4605, - "hfopenllm_v2/BBH": 0.4258, - "hfopenllm_v2/MATH Level 5": 0.0748, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3636, - "hfopenllm_v2/MMLU-PRO": 0.2812 + "hfopenllm_v2/IFEval": 0.4511, + "hfopenllm_v2/BBH": 0.4275, + "hfopenllm_v2/MATH Level 5": 0.1473, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3623, + "hfopenllm_v2/MMLU-PRO": 0.2806 } } ] diff --git a/data/developers/boltmonkey.json b/data/developers/boltmonkey.json index 93a1093c0cbb01b9a84ae22cf6063fc50d8793cb..d93c347b80733b690b666d77f058a8e36708eee2 100644 --- a/data/developers/boltmonkey.json +++ b/data/developers/boltmonkey.json @@ -21,12 +21,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7999, - "hfopenllm_v2/BBH": 0.5152, - "hfopenllm_v2/MATH Level 5": 0.1193, - "hfopenllm_v2/GPQA": 0.281, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3733 + "hfopenllm_v2/IFEval": 0.459, + "hfopenllm_v2/BBH": 0.5185, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.2743, + "hfopenllm_v2/MUSR": 0.4083, + "hfopenllm_v2/MMLU-PRO": 0.3631 } }, { diff --git a/data/developers/bunnycore.json b/data/developers/bunnycore.json index 069b8a9f0214975725e28f414b82b07e73072c0f..153064e2dea6c3626cb6deb17e2d3191b25ec406 100644 --- a/data/developers/bunnycore.json +++ b/data/developers/bunnycore.json @@ -287,12 +287,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4652, - "hfopenllm_v2/BBH": 0.4531, - "hfopenllm_v2/MATH Level 5": 0.1284, - "hfopenllm_v2/GPQA": 0.2643, - "hfopenllm_v2/MUSR": 0.3394, - "hfopenllm_v2/MMLU-PRO": 0.3152 + "hfopenllm_v2/IFEval": 0.1775, + "hfopenllm_v2/BBH": 0.295, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1049 } }, { diff --git a/data/developers/cir-ams.json b/data/developers/cir-ams.json index 09d0cc390a2584b96dbfe9a5acec25171fe175fa..df9dcecb6f8fae5d901c2496b58813341797427e 100644 --- a/data/developers/cir-ams.json +++ b/data/developers/cir-ams.json @@ -7,17 +7,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5736, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Safety": 0.7178, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029, + "reward-bench/Score": 0.8172, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, + "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527 + "reward-bench/Ties": 0.6527, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029 } } ] diff --git a/data/developers/cpayne1303.json b/data/developers/cpayne1303.json index 878ab50b2c306395a2d661382c84d0660b0f0d51..6d735bd94a67b9fc86d407e5a74d4ec119a21a01 100644 --- a/data/developers/cpayne1303.json +++ b/data/developers/cpayne1303.json @@ -35,12 +35,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1949, - "hfopenllm_v2/BBH": 0.2965, - "hfopenllm_v2/MATH Level 5": 0.0045, + "hfopenllm_v2/IFEval": 0.1916, + "hfopenllm_v2/BBH": 0.2977, + "hfopenllm_v2/MATH Level 5": 0.0, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3885, - "hfopenllm_v2/MMLU-PRO": 0.1111 + "hfopenllm_v2/MUSR": 0.3872, + "hfopenllm_v2/MMLU-PRO": 0.1132 } }, { diff --git a/data/developers/daemontatox.json b/data/developers/daemontatox.json index 09a88a4f4f23ea30a1c40f4893201987d0954b30..3de1c87bc255ec29e11a7fcd9434ebc4d17ff27a 100644 --- a/data/developers/daemontatox.json +++ b/data/developers/daemontatox.json @@ -231,12 +231,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3745, - "hfopenllm_v2/BBH": 0.6668, - "hfopenllm_v2/MATH Level 5": 0.4758, - "hfopenllm_v2/GPQA": 0.3943, - "hfopenllm_v2/MUSR": 0.4858, - "hfopenllm_v2/MMLU-PRO": 0.5593 + "hfopenllm_v2/IFEval": 0.4855, + "hfopenllm_v2/BBH": 0.6627, + "hfopenllm_v2/MATH Level 5": 0.4841, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4256, + "hfopenllm_v2/MMLU-PRO": 0.5542 } }, { diff --git a/data/developers/davielion.json b/data/developers/davielion.json index d2a3ca1471ea06bca721c18d505591908b22252e..5027340a9fce3c0b5162453c9d10516ec45234d2 100644 --- a/data/developers/davielion.json +++ b/data/developers/davielion.json @@ -7,12 +7,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1549, - "hfopenllm_v2/BBH": 0.2937, - "hfopenllm_v2/MATH Level 5": 0.006, - "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/IFEval": 0.1507, + "hfopenllm_v2/BBH": 0.293, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2534, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1128 + "hfopenllm_v2/MMLU-PRO": 0.1125 } }, { diff --git a/data/developers/dfurman.json b/data/developers/dfurman.json index 2947dc3ef503295f24886c787e729305dcebb026..7e28f4da929deb69b4028e08d70bcf7cf516d943 100644 --- a/data/developers/dfurman.json +++ b/data/developers/dfurman.json @@ -35,12 +35,12 @@ "developer": "dfurman", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2835, - "hfopenllm_v2/BBH": 0.3842, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3566, - "hfopenllm_v2/MMLU-PRO": 0.2298 + "hfopenllm_v2/IFEval": 0.3, + "hfopenllm_v2/BBH": 0.3853, + "hfopenllm_v2/MATH Level 5": 0.0415, + "hfopenllm_v2/GPQA": 0.2617, + "hfopenllm_v2/MUSR": 0.3579, + "hfopenllm_v2/MMLU-PRO": 0.2281 } }, { diff --git a/data/developers/doppelreflex.json b/data/developers/doppelreflex.json index 0b78fc1cbec001d1df6b60b1fab265b2eab6799e..4e478be7761570eec41b8a497db82c0ab081c8b0 100644 --- a/data/developers/doppelreflex.json +++ b/data/developers/doppelreflex.json @@ -175,12 +175,12 @@ "developer": "DoppelReflEx", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.436, - "hfopenllm_v2/BBH": 0.4956, - "hfopenllm_v2/MATH Level 5": 0.0589, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.3843, - "hfopenllm_v2/MMLU-PRO": 0.3237 + "hfopenllm_v2/IFEval": 0.451, + "hfopenllm_v2/BBH": 0.4944, + "hfopenllm_v2/MATH Level 5": 0.1156, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.3896, + "hfopenllm_v2/MMLU-PRO": 0.3256 } }, { diff --git a/data/developers/google.json b/data/developers/google.json index d2c89d05cf05ad4bda3e0747fb0e85f41c1968c7..ae9dc7717f1b79909d67dfe1364f96967d93b736 100644 --- a/data/developers/google.json +++ b/data/developers/google.json @@ -76,12 +76,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2207, - "hfopenllm_v2/BBH": 0.4537, - "hfopenllm_v2/MATH Level 5": 0.0008, - "hfopenllm_v2/GPQA": 0.2458, - "hfopenllm_v2/MUSR": 0.422, - "hfopenllm_v2/MMLU-PRO": 0.2142 + "hfopenllm_v2/IFEval": 0.2237, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.4181, + "hfopenllm_v2/MMLU-PRO": 0.2147 } }, { @@ -139,7 +139,6 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -147,6 +146,7 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, + "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -723,7 +723,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 16.9 + "terminal-bench-2.0/terminal-bench-2.0": 17.1 } }, { @@ -823,7 +823,7 @@ "reward-bench/Safety": 0.881, "reward-bench/Focus": 0.805, "reward-bench/Ties": 0.811, - "terminal-bench-2.0/terminal-bench-2.0": 16.4 + "terminal-bench-2.0/terminal-bench-2.0": 19.6 } }, { @@ -870,7 +870,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 56.9 + "terminal-bench-2.0/terminal-bench-2.0": 62.2 } }, { @@ -879,7 +879,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.36, + "appworld_test_normal/appworld/test_normal": 0.13, "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, @@ -900,10 +900,10 @@ "global-mmlu-lite/Yoruba": 0.9425, "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, - "swe-bench/swe-bench": 0.71, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, + "swe-bench/swe-bench": 0.7234, + "tau-bench-2_airline/tau-bench-2/airline": 0.7, "tau-bench-2_retail/tau-bench-2/retail": 0.7805, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 } }, { @@ -1056,12 +1056,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5288, - "hfopenllm_v2/BBH": 0.4178, - "hfopenllm_v2/MATH Level 5": 0.0476, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3728, - "hfopenllm_v2/MMLU-PRO": 0.2467 + "hfopenllm_v2/IFEval": 0.5078, + "hfopenllm_v2/BBH": 0.4226, + "hfopenllm_v2/MATH Level 5": 0.0347, + "hfopenllm_v2/GPQA": 0.2852, + "hfopenllm_v2/MUSR": 0.3964, + "hfopenllm_v2/MMLU-PRO": 0.2578 } }, { diff --git a/data/developers/gunulhona.json b/data/developers/gunulhona.json index 3d63c85d80df83c6632e31f815d8dc78510d19a0..1eba4dc6aed35ea93d90c3c8c2e1a2676805ffa0 100644 --- a/data/developers/gunulhona.json +++ b/data/developers/gunulhona.json @@ -21,12 +21,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } } ] diff --git a/data/developers/hendrydong.json b/data/developers/hendrydong.json index 66d1cb6cae0c438dba7144b9a92c7891857c219e..56e05071d3b06a1289763aa84d052b49b755b3aa 100644 --- a/data/developers/hendrydong.json +++ b/data/developers/hendrydong.json @@ -7,17 +7,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5851, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Safety": 0.6956, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508, + "reward-bench/Score": 0.7847, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, + "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988 + "reward-bench/Ties": 0.5988, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508 } } ] diff --git a/data/developers/huggingfacetb.json b/data/developers/huggingfacetb.json index 912c5d4a2004a890736a0d6051d6fabe190615d4..60e6e856fd10ebf29a5d127dd18301cd9edf10dd 100644 --- a/data/developers/huggingfacetb.json +++ b/data/developers/huggingfacetb.json @@ -133,12 +133,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2883, - "hfopenllm_v2/BBH": 0.3124, - "hfopenllm_v2/MATH Level 5": 0.003, - "hfopenllm_v2/GPQA": 0.2357, - "hfopenllm_v2/MUSR": 0.3662, - "hfopenllm_v2/MMLU-PRO": 0.1115 + "hfopenllm_v2/IFEval": 0.0593, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0144, + "hfopenllm_v2/GPQA": 0.2341, + "hfopenllm_v2/MUSR": 0.3871, + "hfopenllm_v2/MMLU-PRO": 0.1092 } }, { diff --git a/data/developers/icefog72.json b/data/developers/icefog72.json index 7e3fc1f7a7c9a270611dd8a76e50d06bd5496253..0ea5ae912613ba6349dc7eed95da5f8d3afd5dbe 100644 --- a/data/developers/icefog72.json +++ b/data/developers/icefog72.json @@ -813,6 +813,20 @@ "hfopenllm_v2/MMLU-PRO": 0.3103 } }, + { + "id": "icefog72/IceSakeV6RP-7b", + "name": "IceSakeV6RP-7b", + "developer": "icefog72", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5033, + "hfopenllm_v2/BBH": 0.4976, + "hfopenllm_v2/MATH Level 5": 0.0619, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.42, + "hfopenllm_v2/MMLU-PRO": 0.3093 + } + }, { "id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", diff --git a/data/developers/internlm.json b/data/developers/internlm.json index 3995d91bd1934deb8a0118739d074676cf30299b..ba5efe87a4d9e153aadeeeaadd0b465b5955bdd6 100644 --- a/data/developers/internlm.json +++ b/data/developers/internlm.json @@ -21,16 +21,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8217, + "reward-bench/Score": 0.3902, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Safety": 0.4711, + "reward-bench/Reasoning": 0.8724, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, - "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Reasoning": 0.8724 + "reward-bench/Ties": 0.1934 } }, { @@ -39,16 +39,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5628, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Safety": 0.6111, - "reward-bench/Reasoning": 0.9576, + "reward-bench/Score": 0.9016, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, + "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483 + "reward-bench/Ties": 0.5483, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Reasoning": 0.9576 } }, { @@ -71,16 +71,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8759, + "reward-bench/Score": 0.5335, + "reward-bench/Chat": 0.9916, + "reward-bench/Chat Hard": 0.6952, + "reward-bench/Safety": 0.5956, + "reward-bench/Reasoning": 0.9453, "reward-bench/Factuality": 0.4211, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5628, - "reward-bench/Safety": 0.8716, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.5164, - "reward-bench/Chat": 0.9916, - "reward-bench/Chat Hard": 0.6952, - "reward-bench/Reasoning": 0.9453 + "reward-bench/Ties": 0.5164 } }, { diff --git a/data/developers/isaak-carter.json b/data/developers/isaak-carter.json index 5e4243cedab9b62fa1fd1f8e9be8ea7f6c708259..690871bc9bb32834a871ee3cfe82d0488b7c6bad 100644 --- a/data/developers/isaak-carter.json +++ b/data/developers/isaak-carter.json @@ -35,12 +35,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2553, - "hfopenllm_v2/BBH": 0.4725, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.3654, - "hfopenllm_v2/MMLU-PRO": 0.3316 + "hfopenllm_v2/IFEval": 0.2477, + "hfopenllm_v2/BBH": 0.4758, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.3641, + "hfopenllm_v2/MMLU-PRO": 0.3292 } } ] diff --git a/data/developers/leroydyer.json b/data/developers/leroydyer.json index 7fdecb6a1e65adfe5948ae6f60f4bc236a803ac7..e1aa95462e1fed8d4e011735c7846112031a21a1 100644 --- a/data/developers/leroydyer.json +++ b/data/developers/leroydyer.json @@ -679,12 +679,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3066, - "hfopenllm_v2/BBH": 0.4577, + "hfopenllm_v2/IFEval": 0.3036, + "hfopenllm_v2/BBH": 0.4575, "hfopenllm_v2/MATH Level 5": 0.0446, - "hfopenllm_v2/GPQA": 0.2995, - "hfopenllm_v2/MUSR": 0.4254, - "hfopenllm_v2/MMLU-PRO": 0.2318 + "hfopenllm_v2/GPQA": 0.3012, + "hfopenllm_v2/MUSR": 0.4253, + "hfopenllm_v2/MMLU-PRO": 0.2329 } }, { diff --git a/data/developers/llmat.json b/data/developers/llmat.json index d073eb81547c22e04e5363702192b3a9654d7362..95633d3199310803501c256677a0fb788d0a08f7 100644 --- a/data/developers/llmat.json +++ b/data/developers/llmat.json @@ -7,12 +7,12 @@ "developer": "llmat", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.377, - "hfopenllm_v2/BBH": 0.3978, - "hfopenllm_v2/MATH Level 5": 0.0242, - "hfopenllm_v2/GPQA": 0.2668, - "hfopenllm_v2/MUSR": 0.3555, - "hfopenllm_v2/MMLU-PRO": 0.2278 + "hfopenllm_v2/IFEval": 0.364, + "hfopenllm_v2/BBH": 0.4005, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2693, + "hfopenllm_v2/MUSR": 0.3529, + "hfopenllm_v2/MMLU-PRO": 0.2301 } } ] diff --git a/data/developers/microsoft.json b/data/developers/microsoft.json index ac7ddab4c9378bc179c9df1d82dca24687067465..5b22c3abc624b7d65a825afffb8ded406b5eb203 100644 --- a/data/developers/microsoft.json +++ b/data/developers/microsoft.json @@ -225,12 +225,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5613, - "hfopenllm_v2/BBH": 0.5676, - "hfopenllm_v2/MATH Level 5": 0.1163, - "hfopenllm_v2/GPQA": 0.3196, - "hfopenllm_v2/MUSR": 0.395, - "hfopenllm_v2/MMLU-PRO": 0.3866 + "hfopenllm_v2/IFEval": 0.5477, + "hfopenllm_v2/BBH": 0.5491, + "hfopenllm_v2/MATH Level 5": 0.1639, + "hfopenllm_v2/GPQA": 0.3322, + "hfopenllm_v2/MUSR": 0.4284, + "hfopenllm_v2/MMLU-PRO": 0.4022 } }, { @@ -341,12 +341,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0488, - "hfopenllm_v2/BBH": 0.6703, - "hfopenllm_v2/MATH Level 5": 0.2787, - "hfopenllm_v2/GPQA": 0.401, + "hfopenllm_v2/IFEval": 0.0585, + "hfopenllm_v2/BBH": 0.6691, + "hfopenllm_v2/MATH Level 5": 0.3165, + "hfopenllm_v2/GPQA": 0.406, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5295 + "hfopenllm_v2/MMLU-PRO": 0.5287 } }, { diff --git a/data/developers/mistralai.json b/data/developers/mistralai.json index 14d2f60be9e807210d185f950d2f81080246d7d7..57d196d824fe3b2d3e0a8303970991b9f16872ee 100644 --- a/data/developers/mistralai.json +++ b/data/developers/mistralai.json @@ -718,12 +718,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2326, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.4413, - "hfopenllm_v2/MMLU-PRO": 0.3871 + "hfopenllm_v2/IFEval": 0.2415, + "hfopenllm_v2/BBH": 0.5087, + "hfopenllm_v2/MATH Level 5": 0.102, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4321, + "hfopenllm_v2/MMLU-PRO": 0.385 } }, { diff --git a/data/developers/mlabonne.json b/data/developers/mlabonne.json index be86bd7fe732025f133b06dc7c412aa5e56b7119..2620a8c4e8931697abdcd44e4a4aae7c1e430da5 100644 --- a/data/developers/mlabonne.json +++ b/data/developers/mlabonne.json @@ -161,12 +161,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4162, - "hfopenllm_v2/BBH": 0.5124, - "hfopenllm_v2/MATH Level 5": 0.0853, - "hfopenllm_v2/GPQA": 0.3029, - "hfopenllm_v2/MUSR": 0.415, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.7561, + "hfopenllm_v2/BBH": 0.5111, + "hfopenllm_v2/MATH Level 5": 0.0906, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3841 } }, { diff --git a/data/developers/multiple.json b/data/developers/multiple.json index e235ffc0287578be5fd9fdf3ba4e4e1b232b5df8..3ca416f37a8a2d37629452327cce332c02e2b948 100644 --- a/data/developers/multiple.json +++ b/data/developers/multiple.json @@ -7,7 +7,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 71.0 + "terminal-bench-2.0/terminal-bench-2.0": 58.4 } } ] diff --git a/data/developers/nazimali.json b/data/developers/nazimali.json index 34d47c9647d462b17a0fe6015c5c4f9fc00264e7..07c42beb48b2fc9e0ad1a1059178a9ce4071a591 100644 --- a/data/developers/nazimali.json +++ b/data/developers/nazimali.json @@ -21,12 +21,12 @@ "developer": "nazimali", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.486, - "hfopenllm_v2/BBH": 0.4721, - "hfopenllm_v2/MATH Level 5": 0.0846, - "hfopenllm_v2/GPQA": 0.2844, - "hfopenllm_v2/MUSR": 0.4006, - "hfopenllm_v2/MMLU-PRO": 0.3087 + "hfopenllm_v2/IFEval": 0.4964, + "hfopenllm_v2/BBH": 0.4699, + "hfopenllm_v2/MATH Level 5": 0.0045, + "hfopenllm_v2/GPQA": 0.2827, + "hfopenllm_v2/MUSR": 0.3979, + "hfopenllm_v2/MMLU-PRO": 0.3063 } } ] diff --git a/data/developers/nexusflow.json b/data/developers/nexusflow.json index 2f78cab8780e463261ce35ef251c295eb3f3fd0f..49fe739e7104f59b148ecb24e067de8a0cb500b5 100644 --- a/data/developers/nexusflow.json +++ b/data/developers/nexusflow.json @@ -21,17 +21,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8133, + "reward-bench/Score": 0.4553, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7556, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137 + "reward-bench/Ties": 0.1004 } } ] diff --git a/data/developers/nicolinho.json b/data/developers/nicolinho.json index 551d4a5de698babd0e830b509f51bb11f4dd2ac7..bf9706fe4ed328188b0945860efeebf7183abc65 100644 --- a/data/developers/nicolinho.json +++ b/data/developers/nicolinho.json @@ -51,16 +51,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9314, + "reward-bench/Score": 0.7074, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Safety": 0.9467, + "reward-bench/Reasoning": 0.9677, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, - "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Reasoning": 0.9677 + "reward-bench/Ties": 0.7234 } } ] diff --git a/data/developers/ontocord.json b/data/developers/ontocord.json index c16f4cbacdda3485b721f459b079923a6793a670..add41dbe183800ebed90c916e0b704c974dd50e7 100644 --- a/data/developers/ontocord.json +++ b/data/developers/ontocord.json @@ -273,12 +273,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1128, - "hfopenllm_v2/BBH": 0.3171, - "hfopenllm_v2/MATH Level 5": 0.0113, - "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.346, - "hfopenllm_v2/MMLU-PRO": 0.1129 + "hfopenllm_v2/IFEval": 0.1162, + "hfopenllm_v2/BBH": 0.3184, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2634, + "hfopenllm_v2/MUSR": 0.3447, + "hfopenllm_v2/MMLU-PRO": 0.1124 } }, { diff --git a/data/developers/openai.json b/data/developers/openai.json index 1d144bd538c9cee7f94220b8970cc16ed56fa6ae..ede5404bc51db07d8e63c7bbcb6be1b005184ac0 100644 --- a/data/developers/openai.json +++ b/data/developers/openai.json @@ -163,16 +163,16 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.515, - "ace/Food Score": 0.65, - "ace/Gaming Score": 0.578, "apex-agents/Overall Pass@1": 0.23, "apex-agents/Overall Pass@8": 0.4, "apex-agents/Overall Mean Score": 0.387, "apex-agents/Investment Banking Pass@1": 0.273, "apex-agents/Management Consulting Pass@1": 0.227, "apex-agents/Corporate Law Pass@1": 0.189, - "apex-agents/Corporate Lawyer Mean Score": 0.443 + "apex-agents/Corporate Lawyer Mean Score": 0.443, + "ace/Overall Score": 0.515, + "ace/Food Score": 0.65, + "ace/Gaming Score": 0.578 } }, { @@ -772,16 +772,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.6493, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Safety": 0.8619, - "reward-bench/Reasoning": 0.8661, + "reward-bench/Score": 0.8673, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, + "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819 + "reward-bench/Ties": 0.7819, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Reasoning": 0.8661 } }, { @@ -859,16 +859,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.5796, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8374, + "reward-bench/Score": 0.8007, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962 + "reward-bench/Ties": 0.6962, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Reasoning": 0.8374 } }, { @@ -922,7 +922,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 43.4 + "terminal-bench-2.0/terminal-bench-2.0": 41.3 } }, { @@ -931,7 +931,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 31.9 + "terminal-bench-2.0/terminal-bench-2.0": 22.2 } }, { @@ -954,7 +954,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 + "terminal-bench-2.0/terminal-bench-2.0": 7.9 } }, { @@ -986,7 +986,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 53.5 + "terminal-bench-2.0/terminal-bench-2.0": 36.9 } }, { @@ -1013,7 +1013,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.7 + "terminal-bench-2.0/terminal-bench-2.0": 54.0 } }, { @@ -1023,13 +1023,13 @@ "evaluator_relationship": null, "benchmark_scores": { "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.26, + "browsecompplus/browsecompplus": 0.48, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.57, - "tau-bench-2_airline/tau-bench-2/airline": 0.6, - "tau-bench-2_retail/tau-bench-2/retail": 0.68, + "swe-bench/swe-bench": 0.5455, + "tau-bench-2_airline/tau-bench-2/airline": 0.48, + "tau-bench-2_retail/tau-bench-2/retail": 0.51, "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, @@ -1048,7 +1048,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.6 + "terminal-bench-2.0/terminal-bench-2.0": 77.3 } }, { @@ -1233,9 +1233,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.0143, - "livecodebenchpro/Medium Problems": 0.2923, - "livecodebenchpro/Easy Problems": 0.8571 + "livecodebenchpro/Hard Problems": 0.014084507042253521, + "livecodebenchpro/Medium Problems": 0.30985915492957744, + "livecodebenchpro/Easy Problems": 0.8873239436619719 } }, { diff --git a/data/developers/openassistant.json b/data/developers/openassistant.json index 0d3e3f32693fa46354d7063e0615e0d45e8641c5..95e3fbf7702e838e1924ccdd51ef3a7ea27c9d5c 100644 --- a/data/developers/openassistant.json +++ b/data/developers/openassistant.json @@ -7,17 +7,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.615, + "reward-bench/Score": 0.2653, + "reward-bench/Chat": 0.9246, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.3289, + "reward-bench/Reasoning": 0.5855, + "reward-bench/Prior Sets (0.5 weight)": 0.6801, "reward-bench/Factuality": 0.3979, "reward-bench/Precise IF": 0.2875, "reward-bench/Math": 0.377, - "reward-bench/Safety": 0.5446, "reward-bench/Focus": 0.1535, - "reward-bench/Ties": 0.047, - "reward-bench/Chat": 0.9246, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.5855, - "reward-bench/Prior Sets (0.5 weight)": 0.6801 + "reward-bench/Ties": 0.047 } }, { @@ -26,17 +26,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6901, + "reward-bench/Score": 0.2648, + "reward-bench/Chat": 0.8855, + "reward-bench/Chat Hard": 0.4868, + "reward-bench/Safety": 0.3244, + "reward-bench/Reasoning": 0.7752, + "reward-bench/Prior Sets (0.5 weight)": 0.6533, "reward-bench/Factuality": 0.3179, "reward-bench/Precise IF": 0.2625, "reward-bench/Math": 0.3934, - "reward-bench/Safety": 0.6311, "reward-bench/Focus": 0.2707, - "reward-bench/Ties": 0.0198, - "reward-bench/Chat": 0.8855, - "reward-bench/Chat Hard": 0.4868, - "reward-bench/Reasoning": 0.7752, - "reward-bench/Prior Sets (0.5 weight)": 0.6533 + "reward-bench/Ties": 0.0198 } }, { @@ -59,17 +59,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6126, + "reward-bench/Score": 0.32, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Safety": 0.3667, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, - "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836 + "reward-bench/Ties": 0.12 } } ] diff --git a/data/developers/openbmb.json b/data/developers/openbmb.json index d6a398c44a61b95d325c54de1768af4a13c46f1f..b9795ed819d46987af7a443e489394fa08308cdd 100644 --- a/data/developers/openbmb.json +++ b/data/developers/openbmb.json @@ -21,17 +21,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8159, + "reward-bench/Score": 0.5806, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Safety": 0.6267, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, - "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172 + "reward-bench/Ties": 0.5972 } }, { @@ -68,17 +68,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6903, + "reward-bench/Score": 0.4683, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Safety": 0.5089, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, - "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294 + "reward-bench/Ties": 0.3036 } } ] diff --git a/data/developers/pku-alignment.json b/data/developers/pku-alignment.json index 0ae80803f93980ecd8558a983e84a1646a1bd7d6..4cebe3d4c55c06b2bbbd44b7bb6fe487a030b922 100644 --- a/data/developers/pku-alignment.json +++ b/data/developers/pku-alignment.json @@ -26,17 +26,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4727, + "reward-bench/Score": 0.1606, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Safety": 0.1422, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, - "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993 + "reward-bench/Ties": -0.01 } }, { diff --git a/data/developers/primeintellect.json b/data/developers/primeintellect.json index 674a0e3b141480d7e0d33d0a2fe9b205b710216f..160722785b06f80d9b220dee435fad1245d45495 100644 --- a/data/developers/primeintellect.json +++ b/data/developers/primeintellect.json @@ -8,11 +8,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.276, + "hfopenllm_v2/BBH": 0.274, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3339, - "hfopenllm_v2/MMLU-PRO": 0.1123 + "hfopenllm_v2/GPQA": 0.25, + "hfopenllm_v2/MUSR": 0.3753, + "hfopenllm_v2/MMLU-PRO": 0.112 } }, { diff --git a/data/developers/prithivmlmods.json b/data/developers/prithivmlmods.json index f14f3864a910968b11c756eb8300bd4bd3eea36e..88e743a4279f310dec935aed3968189a78be084e 100644 --- a/data/developers/prithivmlmods.json +++ b/data/developers/prithivmlmods.json @@ -63,12 +63,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6064, - "hfopenllm_v2/BBH": 0.6296, - "hfopenllm_v2/MATH Level 5": 0.3708, - "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4873, - "hfopenllm_v2/MMLU-PRO": 0.5307 + "hfopenllm_v2/IFEval": 0.6052, + "hfopenllm_v2/BBH": 0.6317, + "hfopenllm_v2/MATH Level 5": 0.4789, + "hfopenllm_v2/GPQA": 0.3742, + "hfopenllm_v2/MUSR": 0.486, + "hfopenllm_v2/MMLU-PRO": 0.5302 } }, { diff --git a/data/developers/qingy2019.json b/data/developers/qingy2019.json index f93967ea7ea0a389f1379b7b85428b832b816fb9..b11a9d79a541432fdd0c8e33c56cfaa201540388 100644 --- a/data/developers/qingy2019.json +++ b/data/developers/qingy2019.json @@ -35,12 +35,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2401, - "hfopenllm_v2/BBH": 0.4622, - "hfopenllm_v2/MATH Level 5": 0.0725, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3703, - "hfopenllm_v2/MMLU-PRO": 0.2379 + "hfopenllm_v2/IFEval": 0.2358, + "hfopenllm_v2/BBH": 0.4612, + "hfopenllm_v2/MATH Level 5": 0.0642, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3717, + "hfopenllm_v2/MMLU-PRO": 0.2382 } }, { @@ -49,12 +49,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6066, - "hfopenllm_v2/BBH": 0.635, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.3725, + "hfopenllm_v2/IFEval": 0.6005, + "hfopenllm_v2/BBH": 0.6356, + "hfopenllm_v2/MATH Level 5": 0.2764, + "hfopenllm_v2/GPQA": 0.3691, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5331 + "hfopenllm_v2/MMLU-PRO": 0.5339 } }, { diff --git a/data/developers/quazim0t0.json b/data/developers/quazim0t0.json index 135f1135e53ac7da23b4b9457344787d94d6e680..e5588b4aeabec7b69b53d4145ba5bbe548d82ff9 100644 --- a/data/developers/quazim0t0.json +++ b/data/developers/quazim0t0.json @@ -637,12 +637,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7016, - "hfopenllm_v2/BBH": 0.6942, - "hfopenllm_v2/MATH Level 5": 0.4116, - "hfopenllm_v2/GPQA": 0.3624, - "hfopenllm_v2/MUSR": 0.4571, - "hfopenllm_v2/MMLU-PRO": 0.5411 + "hfopenllm_v2/IFEval": 0.2922, + "hfopenllm_v2/BBH": 0.6559, + "hfopenllm_v2/MATH Level 5": 0.2545, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3929, + "hfopenllm_v2/MMLU-PRO": 0.5207 } }, { diff --git a/data/developers/qwen.json b/data/developers/qwen.json index 727516da8d72881c91634dd8fcd12326550fd1bf..3433447d4af65af4b79f785b8a7b43fc273f9d86 100644 --- a/data/developers/qwen.json +++ b/data/developers/qwen.json @@ -1177,12 +1177,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6101, - "hfopenllm_v2/BBH": 0.5008, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3352 + "hfopenllm_v2/IFEval": 0.6147, + "hfopenllm_v2/BBH": 0.4999, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { diff --git a/data/developers/ray2333.json b/data/developers/ray2333.json index 535b872cfc093237ac0d0dc85564191f8ac0a72c..51f9b51d9ebb192a98a4a9a46633e05cf73f77b6 100644 --- a/data/developers/ray2333.json +++ b/data/developers/ray2333.json @@ -79,17 +79,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.589, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209, + "reward-bench/Score": 0.8464, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743 + "reward-bench/Ties": 0.5743, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209 } }, { @@ -116,17 +116,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6089, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Safety": 0.7867, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309, + "reward-bench/Score": 0.8542, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, + "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981 + "reward-bench/Ties": 0.5981, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309 } }, { diff --git a/data/developers/rombodawg.json b/data/developers/rombodawg.json index a241fbd73dcecc7fc768ef47719ff98c1bcb64ab..e7b347a91d5be7f4c52661e255a50c99caa3a3fd 100644 --- a/data/developers/rombodawg.json +++ b/data/developers/rombodawg.json @@ -133,12 +133,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2595, - "hfopenllm_v2/BBH": 0.3884, - "hfopenllm_v2/MATH Level 5": 0.0914, - "hfopenllm_v2/GPQA": 0.2743, + "hfopenllm_v2/IFEval": 0.2566, + "hfopenllm_v2/BBH": 0.39, + "hfopenllm_v2/MATH Level 5": 0.1208, + "hfopenllm_v2/GPQA": 0.2626, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2719 + "hfopenllm_v2/MMLU-PRO": 0.2741 } }, { diff --git a/data/developers/sfairxc.json b/data/developers/sfairxc.json index f511504fa472f754b653d8fff6432038f2fa5642..0f83a5aa819fcf69906563e63fd5c5a7cc1f8ce0 100644 --- a/data/developers/sfairxc.json +++ b/data/developers/sfairxc.json @@ -7,17 +7,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6292, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492, + "reward-bench/Score": 0.8338, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647 + "reward-bench/Ties": 0.6647, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492 } } ] diff --git a/data/developers/shikaichen.json b/data/developers/shikaichen.json index 6502ef5b0e02cc85862537458f25816eb0826b7d..6162cba6dec7eaed27af88272ffb98342af2522b 100644 --- a/data/developers/shikaichen.json +++ b/data/developers/shikaichen.json @@ -7,16 +7,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7249, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.9903, + "reward-bench/Score": 0.9499, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, + "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633 + "reward-bench/Ties": 0.7633, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Reasoning": 0.9903 } } ] diff --git a/data/developers/skywork.json b/data/developers/skywork.json index 15f843864c8bce10620ff0dcdbde98d85af98ac4..a1886dbde5ab2f08ceeb175a3c92e5b11655f24d 100644 --- a/data/developers/skywork.json +++ b/data/developers/skywork.json @@ -71,16 +71,16 @@ "hfopenllm_v2/GPQA": 0.344, "hfopenllm_v2/MUSR": 0.4231, "hfopenllm_v2/MMLU-PRO": 0.4103, - "reward-bench/Score": 0.9426, + "reward-bench/Score": 0.7531, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.8991, + "reward-bench/Safety": 0.9689, + "reward-bench/Reasoning": 0.9807, "reward-bench/Factuality": 0.7674, "reward-bench/Precise IF": 0.375, "reward-bench/Math": 0.6721, - "reward-bench/Safety": 0.9297, "reward-bench/Focus": 0.9172, - "reward-bench/Ties": 0.8182, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.8991, - "reward-bench/Reasoning": 0.9807 + "reward-bench/Ties": 0.8182 } }, { @@ -89,16 +89,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9252, + "reward-bench/Score": 0.7314, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8728, + "reward-bench/Safety": 0.9333, + "reward-bench/Reasoning": 0.962, "reward-bench/Factuality": 0.6989, "reward-bench/Precise IF": 0.425, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.9616, - "reward-bench/Ties": 0.741, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8728, - "reward-bench/Reasoning": 0.962 + "reward-bench/Ties": 0.741 } }, { diff --git a/data/developers/tanliboy.json b/data/developers/tanliboy.json index 7b17e651b33310ae087a6456408950e1456ff623..daa17945601d8528cbfc7c883e4dc4315ea363fc 100644 --- a/data/developers/tanliboy.json +++ b/data/developers/tanliboy.json @@ -7,12 +7,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1829, - "hfopenllm_v2/BBH": 0.5488, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.4056, - "hfopenllm_v2/MMLU-PRO": 0.3805 + "hfopenllm_v2/IFEval": 0.4501, + "hfopenllm_v2/BBH": 0.5472, + "hfopenllm_v2/MATH Level 5": 0.0944, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4017, + "hfopenllm_v2/MMLU-PRO": 0.3792 } }, { diff --git a/data/developers/valiantlabs.json b/data/developers/valiantlabs.json index b0b3adff2288383cd6923301f5751d53ee8ba504..fea79922ada6f8587a14d0c683f2c0090d2ca061 100644 --- a/data/developers/valiantlabs.json +++ b/data/developers/valiantlabs.json @@ -49,12 +49,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3496, - "hfopenllm_v2/BBH": 0.4947, - "hfopenllm_v2/MATH Level 5": 0.1269, - "hfopenllm_v2/GPQA": 0.3037, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.3644 + "hfopenllm_v2/IFEval": 0.7168, + "hfopenllm_v2/BBH": 0.4911, + "hfopenllm_v2/MATH Level 5": 0.1533, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.3512, + "hfopenllm_v2/MMLU-PRO": 0.3663 } }, { @@ -105,12 +105,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2678, - "hfopenllm_v2/BBH": 0.4429, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.302, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.2927 + "hfopenllm_v2/IFEval": 0.6496, + "hfopenllm_v2/BBH": 0.4774, + "hfopenllm_v2/MATH Level 5": 0.0566, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.3909, + "hfopenllm_v2/MMLU-PRO": 0.3382 } }, { diff --git a/data/developers/xai.json b/data/developers/xai.json index 5bf50e8754b14e71e2153e4564b26c11231d3d27..d0ba30a6d3c372a4bdb6c48d3c0cb90677ed30d5 100644 --- a/data/developers/xai.json +++ b/data/developers/xai.json @@ -78,7 +78,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.1 + "terminal-bench-2.0/terminal-bench-2.0": 25.4 } }, { diff --git a/data/developers/yam-peleg.json b/data/developers/yam-peleg.json index f415128b8e93253c12ae15436b68cf36d5f248bf..3161f95274f1998cddb90c53a59dce4097dbe0fd 100644 --- a/data/developers/yam-peleg.json +++ b/data/developers/yam-peleg.json @@ -35,12 +35,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.177, - "hfopenllm_v2/BBH": 0.3411, - "hfopenllm_v2/MATH Level 5": 0.031, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.374, - "hfopenllm_v2/MMLU-PRO": 0.2529 + "hfopenllm_v2/IFEval": 0.1856, + "hfopenllm_v2/BBH": 0.4149, + "hfopenllm_v2/MATH Level 5": 0.0234, + "hfopenllm_v2/GPQA": 0.276, + "hfopenllm_v2/MUSR": 0.3765, + "hfopenllm_v2/MMLU-PRO": 0.2573 } } ] diff --git a/data/developers/ycros.json b/data/developers/ycros.json index 9b83f7f872f197f98fd71538cda202af618f5659..e95400442aba80bbdb2ffa85224e7399ce946b0d 100644 --- a/data/developers/ycros.json +++ b/data/developers/ycros.json @@ -7,12 +7,12 @@ "developer": "ycros", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5994, - "hfopenllm_v2/BBH": 0.5159, - "hfopenllm_v2/MATH Level 5": 0.0785, - "hfopenllm_v2/GPQA": 0.3045, - "hfopenllm_v2/MUSR": 0.4203, - "hfopenllm_v2/MMLU-PRO": 0.3473 + "hfopenllm_v2/IFEval": 0.6262, + "hfopenllm_v2/BBH": 0.5142, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.3079, + "hfopenllm_v2/MUSR": 0.4138, + "hfopenllm_v2/MMLU-PRO": 0.3481 } } ] diff --git a/data/developers/z-ai.json b/data/developers/z-ai.json index 25c72391d11bee4c4ccdbee1d0c2cdded93b09ab..d4ada9599b117d1b4893fef821933bb44c4fb1d9 100644 --- a/data/developers/z-ai.json +++ b/data/developers/z-ai.json @@ -7,7 +7,7 @@ "developer": "Z-AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 33.4 + "terminal-bench-2.0/terminal-bench-2.0": 33.3 } }, { diff --git a/data/models.json b/data/models.json index a4ab0a1994ca0a7faae8835c098ea436fc0ab99c..b08337869e0315ca25ad74e1e682be0126b52503 100644 --- a/data/models.json +++ b/data/models.json @@ -1005,12 +1005,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1685, - "hfopenllm_v2/BBH": 0.3124, - "hfopenllm_v2/MATH Level 5": 0.0015, - "hfopenllm_v2/GPQA": 0.2492, - "hfopenllm_v2/MUSR": 0.3963, - "hfopenllm_v2/MMLU-PRO": 0.1066 + "hfopenllm_v2/IFEval": 0.1746, + "hfopenllm_v2/BBH": 0.3126, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.245, + "hfopenllm_v2/MUSR": 0.4096, + "hfopenllm_v2/MMLU-PRO": 0.1087 } }, { @@ -1391,10 +1391,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6895, + "reward-bench/Score": 0.7008, "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3706, - "reward-bench/Safety": 0.7595 + "reward-bench/Chat Hard": 0.3882, + "reward-bench/Safety": 0.7757 } }, { @@ -2243,7 +2243,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 25.4 + "terminal-bench-2.0/terminal-bench-2.0": 23.9 } }, { @@ -2390,17 +2390,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7606, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Safety": 0.8844, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.9021, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835 + "reward-bench/Ties": 0.8835, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -2508,12 +2508,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { @@ -2536,17 +2536,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.687, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Safety": 0.86, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8431, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397 + "reward-bench/Ties": 0.6397, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -2609,17 +2609,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8551, + "reward-bench/Score": 0.6821, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.7917, + "reward-bench/Safety": 0.8978, + "reward-bench/Reasoning": 0.8005, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7326, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8784, "reward-bench/Focus": 0.8889, - "reward-bench/Ties": 0.6063, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.7917, - "reward-bench/Reasoning": 0.8005, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6063 } }, { @@ -3381,6 +3381,21 @@ "reward-bench/Ties": 0.3534 } }, + { + "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "developer": "allenai", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.5151, + "reward-bench/Factuality": 0.6484, + "reward-bench/Precise IF": 0.3312, + "reward-bench/Math": 0.5574, + "reward-bench/Safety": 0.7289, + "reward-bench/Focus": 0.4889, + "reward-bench/Ties": 0.3357 + } + }, { "id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", @@ -6541,11 +6556,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1842, - "hfopenllm_v2/BBH": 0.2974, - "hfopenllm_v2/MATH Level 5": 0.0053, - "hfopenllm_v2/GPQA": 0.2525, - "hfopenllm_v2/MUSR": 0.378, + "hfopenllm_v2/IFEval": 0.1918, + "hfopenllm_v2/BBH": 0.2969, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/MUSR": 0.3846, "hfopenllm_v2/MMLU-PRO": 0.1169 } }, @@ -6985,16 +7000,16 @@ "helm_mmlu/Virology": 0.602, "helm_mmlu/World Religions": 0.924, "helm_mmlu/Mean win rate": 0.17, - "reward-bench/Score": 0.6466, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.7401, - "reward-bench/Safety": 0.8519, - "reward-bench/Reasoning": 0.8469, + "reward-bench/Score": 0.8417, "reward-bench/Factuality": 0.5284, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.8697, - "reward-bench/Ties": 0.674 + "reward-bench/Ties": 0.674, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.7401, + "reward-bench/Reasoning": 0.8469 } }, { @@ -7152,17 +7167,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.7289, + "reward-bench/Score": 0.3711, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Safety": 0.595, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, - "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635 + "reward-bench/Ties": 0.0899 } }, { @@ -7217,16 +7232,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.8008, + "reward-bench/Score": 0.5744, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.8378, + "reward-bench/Reasoning": 0.7868, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Reasoning": 0.7868 + "reward-bench/Ties": 0.5601 } }, { @@ -7306,7 +7321,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.5 + "terminal-bench-2.0/terminal-bench-2.0": 35.5 } }, { @@ -7431,12 +7446,12 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { + "browsecompplus/browsecompplus": 0.49, "appworld_test_normal/appworld/test_normal": 0.64, - "browsecompplus/browsecompplus": 0.61, - "swe-bench/swe-bench": 0.6061, - "tau-bench-2_airline/tau-bench-2/airline": 0.72, + "swe-bench/swe-bench": 0.8072, + "tau-bench-2_airline/tau-bench-2/airline": 0.66, "tau-bench-2_retail/tau-bench-2/retail": 0.85, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } }, { @@ -7445,7 +7460,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 34.8 + "terminal-bench-2.0/terminal-bench-2.0": 38.0 } }, { @@ -7454,7 +7469,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 52.1 + "terminal-bench-2.0/terminal-bench-2.0": 59.1 } }, { @@ -7463,7 +7478,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.7 + "terminal-bench-2.0/terminal-bench-2.0": 66.9 } }, { @@ -7537,7 +7552,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.5 + "terminal-bench-2.0/terminal-bench-2.0": 40.1 } }, { @@ -8076,12 +8091,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4605, - "hfopenllm_v2/BBH": 0.4258, - "hfopenllm_v2/MATH Level 5": 0.0748, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3636, - "hfopenllm_v2/MMLU-PRO": 0.2812 + "hfopenllm_v2/IFEval": 0.4511, + "hfopenllm_v2/BBH": 0.4275, + "hfopenllm_v2/MATH Level 5": 0.1473, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3623, + "hfopenllm_v2/MMLU-PRO": 0.2806 } }, { @@ -9828,12 +9843,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7999, - "hfopenllm_v2/BBH": 0.5152, - "hfopenllm_v2/MATH Level 5": 0.1193, - "hfopenllm_v2/GPQA": 0.281, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3733 + "hfopenllm_v2/IFEval": 0.459, + "hfopenllm_v2/BBH": 0.5185, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.2743, + "hfopenllm_v2/MUSR": 0.4083, + "hfopenllm_v2/MMLU-PRO": 0.3631 } }, { @@ -10584,12 +10599,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4652, - "hfopenllm_v2/BBH": 0.4531, - "hfopenllm_v2/MATH Level 5": 0.1284, - "hfopenllm_v2/GPQA": 0.2643, - "hfopenllm_v2/MUSR": 0.3394, - "hfopenllm_v2/MMLU-PRO": 0.3152 + "hfopenllm_v2/IFEval": 0.1775, + "hfopenllm_v2/BBH": 0.295, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1049 } }, { @@ -11813,17 +11828,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5736, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Safety": 0.7178, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029, + "reward-bench/Score": 0.8172, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, + "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527 + "reward-bench/Ties": 0.6527, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029 } }, { @@ -13385,12 +13400,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1949, - "hfopenllm_v2/BBH": 0.2965, - "hfopenllm_v2/MATH Level 5": 0.0045, + "hfopenllm_v2/IFEval": 0.1916, + "hfopenllm_v2/BBH": 0.2977, + "hfopenllm_v2/MATH Level 5": 0.0, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3885, - "hfopenllm_v2/MMLU-PRO": 0.1111 + "hfopenllm_v2/MUSR": 0.3872, + "hfopenllm_v2/MMLU-PRO": 0.1132 } }, { @@ -14323,12 +14338,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3745, - "hfopenllm_v2/BBH": 0.6668, - "hfopenllm_v2/MATH Level 5": 0.4758, - "hfopenllm_v2/GPQA": 0.3943, - "hfopenllm_v2/MUSR": 0.4858, - "hfopenllm_v2/MMLU-PRO": 0.5593 + "hfopenllm_v2/IFEval": 0.4855, + "hfopenllm_v2/BBH": 0.6627, + "hfopenllm_v2/MATH Level 5": 0.4841, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4256, + "hfopenllm_v2/MMLU-PRO": 0.5542 } }, { @@ -15378,12 +15393,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1549, - "hfopenllm_v2/BBH": 0.2937, - "hfopenllm_v2/MATH Level 5": 0.006, - "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/IFEval": 0.1507, + "hfopenllm_v2/BBH": 0.293, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2534, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1128 + "hfopenllm_v2/MMLU-PRO": 0.1125 } }, { @@ -16361,12 +16376,12 @@ "developer": "dfurman", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2835, - "hfopenllm_v2/BBH": 0.3842, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3566, - "hfopenllm_v2/MMLU-PRO": 0.2298 + "hfopenllm_v2/IFEval": 0.3, + "hfopenllm_v2/BBH": 0.3853, + "hfopenllm_v2/MATH Level 5": 0.0415, + "hfopenllm_v2/GPQA": 0.2617, + "hfopenllm_v2/MUSR": 0.3579, + "hfopenllm_v2/MMLU-PRO": 0.2281 } }, { @@ -17005,12 +17020,12 @@ "developer": "DoppelReflEx", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.436, - "hfopenllm_v2/BBH": 0.4956, - "hfopenllm_v2/MATH Level 5": 0.0589, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.3843, - "hfopenllm_v2/MMLU-PRO": 0.3237 + "hfopenllm_v2/IFEval": 0.451, + "hfopenllm_v2/BBH": 0.4944, + "hfopenllm_v2/MATH Level 5": 0.1156, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.3896, + "hfopenllm_v2/MMLU-PRO": 0.3256 } }, { @@ -23441,12 +23456,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2207, - "hfopenllm_v2/BBH": 0.4537, - "hfopenllm_v2/MATH Level 5": 0.0008, - "hfopenllm_v2/GPQA": 0.2458, - "hfopenllm_v2/MUSR": 0.422, - "hfopenllm_v2/MMLU-PRO": 0.2142 + "hfopenllm_v2/IFEval": 0.2237, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.4181, + "hfopenllm_v2/MMLU-PRO": 0.2147 } }, { @@ -23504,7 +23519,6 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -23512,6 +23526,7 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, + "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -24088,7 +24103,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 16.9 + "terminal-bench-2.0/terminal-bench-2.0": 17.1 } }, { @@ -24188,7 +24203,7 @@ "reward-bench/Safety": 0.881, "reward-bench/Focus": 0.805, "reward-bench/Ties": 0.811, - "terminal-bench-2.0/terminal-bench-2.0": 16.4 + "terminal-bench-2.0/terminal-bench-2.0": 19.6 } }, { @@ -24235,7 +24250,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 56.9 + "terminal-bench-2.0/terminal-bench-2.0": 62.2 } }, { @@ -24244,7 +24259,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.36, + "appworld_test_normal/appworld/test_normal": 0.13, "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, @@ -24265,10 +24280,10 @@ "global-mmlu-lite/Yoruba": 0.9425, "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, - "swe-bench/swe-bench": 0.71, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, + "swe-bench/swe-bench": 0.7234, + "tau-bench-2_airline/tau-bench-2/airline": 0.7, "tau-bench-2_retail/tau-bench-2/retail": 0.7805, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 } }, { @@ -24421,12 +24436,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5288, - "hfopenllm_v2/BBH": 0.4178, - "hfopenllm_v2/MATH Level 5": 0.0476, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3728, - "hfopenllm_v2/MMLU-PRO": 0.2467 + "hfopenllm_v2/IFEval": 0.5078, + "hfopenllm_v2/BBH": 0.4226, + "hfopenllm_v2/MATH Level 5": 0.0347, + "hfopenllm_v2/GPQA": 0.2852, + "hfopenllm_v2/MUSR": 0.3964, + "hfopenllm_v2/MMLU-PRO": 0.2578 } }, { @@ -25543,12 +25558,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } }, { @@ -25879,17 +25894,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5851, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Safety": 0.6956, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508, + "reward-bench/Score": 0.7847, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, + "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988 + "reward-bench/Ties": 0.5988, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508 } }, { @@ -26771,12 +26786,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2883, - "hfopenllm_v2/BBH": 0.3124, - "hfopenllm_v2/MATH Level 5": 0.003, - "hfopenllm_v2/GPQA": 0.2357, - "hfopenllm_v2/MUSR": 0.3662, - "hfopenllm_v2/MMLU-PRO": 0.1115 + "hfopenllm_v2/IFEval": 0.0593, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0144, + "hfopenllm_v2/GPQA": 0.2341, + "hfopenllm_v2/MUSR": 0.3871, + "hfopenllm_v2/MMLU-PRO": 0.1092 } }, { @@ -28206,6 +28221,20 @@ "hfopenllm_v2/MMLU-PRO": 0.3103 } }, + { + "id": "icefog72/IceSakeV6RP-7b", + "name": "IceSakeV6RP-7b", + "developer": "icefog72", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5033, + "hfopenllm_v2/BBH": 0.4976, + "hfopenllm_v2/MATH Level 5": 0.0619, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.42, + "hfopenllm_v2/MMLU-PRO": 0.3093 + } + }, { "id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", @@ -28622,16 +28651,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8217, + "reward-bench/Score": 0.3902, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Safety": 0.4711, + "reward-bench/Reasoning": 0.8724, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, - "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Reasoning": 0.8724 + "reward-bench/Ties": 0.1934 } }, { @@ -28640,16 +28669,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5628, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Safety": 0.6111, - "reward-bench/Reasoning": 0.9576, + "reward-bench/Score": 0.9016, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, + "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483 + "reward-bench/Ties": 0.5483, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Reasoning": 0.9576 } }, { @@ -28672,16 +28701,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8759, + "reward-bench/Score": 0.5335, + "reward-bench/Chat": 0.9916, + "reward-bench/Chat Hard": 0.6952, + "reward-bench/Safety": 0.5956, + "reward-bench/Reasoning": 0.9453, "reward-bench/Factuality": 0.4211, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5628, - "reward-bench/Safety": 0.8716, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.5164, - "reward-bench/Chat": 0.9916, - "reward-bench/Chat Hard": 0.6952, - "reward-bench/Reasoning": 0.9453 + "reward-bench/Ties": 0.5164 } }, { @@ -28956,12 +28985,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2553, - "hfopenllm_v2/BBH": 0.4725, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.3654, - "hfopenllm_v2/MMLU-PRO": 0.3316 + "hfopenllm_v2/IFEval": 0.2477, + "hfopenllm_v2/BBH": 0.4758, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.3641, + "hfopenllm_v2/MMLU-PRO": 0.3292 } }, { @@ -37955,12 +37984,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3066, - "hfopenllm_v2/BBH": 0.4577, + "hfopenllm_v2/IFEval": 0.3036, + "hfopenllm_v2/BBH": 0.4575, "hfopenllm_v2/MATH Level 5": 0.0446, - "hfopenllm_v2/GPQA": 0.2995, - "hfopenllm_v2/MUSR": 0.4254, - "hfopenllm_v2/MMLU-PRO": 0.2318 + "hfopenllm_v2/GPQA": 0.3012, + "hfopenllm_v2/MUSR": 0.4253, + "hfopenllm_v2/MMLU-PRO": 0.2329 } }, { @@ -38907,12 +38936,12 @@ "developer": "llmat", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.377, - "hfopenllm_v2/BBH": 0.3978, - "hfopenllm_v2/MATH Level 5": 0.0242, - "hfopenllm_v2/GPQA": 0.2668, - "hfopenllm_v2/MUSR": 0.3555, - "hfopenllm_v2/MMLU-PRO": 0.2278 + "hfopenllm_v2/IFEval": 0.364, + "hfopenllm_v2/BBH": 0.4005, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2693, + "hfopenllm_v2/MUSR": 0.3529, + "hfopenllm_v2/MMLU-PRO": 0.2301 } }, { @@ -43206,12 +43235,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5613, - "hfopenllm_v2/BBH": 0.5676, - "hfopenllm_v2/MATH Level 5": 0.1163, - "hfopenllm_v2/GPQA": 0.3196, - "hfopenllm_v2/MUSR": 0.395, - "hfopenllm_v2/MMLU-PRO": 0.3866 + "hfopenllm_v2/IFEval": 0.5477, + "hfopenllm_v2/BBH": 0.5491, + "hfopenllm_v2/MATH Level 5": 0.1639, + "hfopenllm_v2/GPQA": 0.3322, + "hfopenllm_v2/MUSR": 0.4284, + "hfopenllm_v2/MMLU-PRO": 0.4022 } }, { @@ -43322,12 +43351,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0488, - "hfopenllm_v2/BBH": 0.6703, - "hfopenllm_v2/MATH Level 5": 0.2787, - "hfopenllm_v2/GPQA": 0.401, + "hfopenllm_v2/IFEval": 0.0585, + "hfopenllm_v2/BBH": 0.6691, + "hfopenllm_v2/MATH Level 5": 0.3165, + "hfopenllm_v2/GPQA": 0.406, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5295 + "hfopenllm_v2/MMLU-PRO": 0.5287 } }, { @@ -44423,12 +44452,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2326, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.4413, - "hfopenllm_v2/MMLU-PRO": 0.3871 + "hfopenllm_v2/IFEval": 0.2415, + "hfopenllm_v2/BBH": 0.5087, + "hfopenllm_v2/MATH Level 5": 0.102, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4321, + "hfopenllm_v2/MMLU-PRO": 0.385 } }, { @@ -44729,12 +44758,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4162, - "hfopenllm_v2/BBH": 0.5124, - "hfopenllm_v2/MATH Level 5": 0.0853, - "hfopenllm_v2/GPQA": 0.3029, - "hfopenllm_v2/MUSR": 0.415, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.7561, + "hfopenllm_v2/BBH": 0.5111, + "hfopenllm_v2/MATH Level 5": 0.0906, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3841 } }, { @@ -45288,7 +45317,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 71.0 + "terminal-bench-2.0/terminal-bench-2.0": 58.4 } }, { @@ -45604,12 +45633,12 @@ "developer": "nazimali", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.486, - "hfopenllm_v2/BBH": 0.4721, - "hfopenllm_v2/MATH Level 5": 0.0846, - "hfopenllm_v2/GPQA": 0.2844, - "hfopenllm_v2/MUSR": 0.4006, - "hfopenllm_v2/MMLU-PRO": 0.3087 + "hfopenllm_v2/IFEval": 0.4964, + "hfopenllm_v2/BBH": 0.4699, + "hfopenllm_v2/MATH Level 5": 0.0045, + "hfopenllm_v2/GPQA": 0.2827, + "hfopenllm_v2/MUSR": 0.3979, + "hfopenllm_v2/MMLU-PRO": 0.3063 } }, { @@ -48113,17 +48142,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8133, + "reward-bench/Score": 0.4553, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7556, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137 + "reward-bench/Ties": 0.1004 } }, { @@ -48288,16 +48317,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9314, + "reward-bench/Score": 0.7074, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Safety": 0.9467, + "reward-bench/Reasoning": 0.9677, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, - "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Reasoning": 0.9677 + "reward-bench/Ties": 0.7234 } }, { @@ -50364,12 +50393,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1128, - "hfopenllm_v2/BBH": 0.3171, - "hfopenllm_v2/MATH Level 5": 0.0113, - "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.346, - "hfopenllm_v2/MMLU-PRO": 0.1129 + "hfopenllm_v2/IFEval": 0.1162, + "hfopenllm_v2/BBH": 0.3184, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2634, + "hfopenllm_v2/MUSR": 0.3447, + "hfopenllm_v2/MMLU-PRO": 0.1124 } }, { @@ -50982,16 +51011,16 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.515, - "ace/Food Score": 0.65, - "ace/Gaming Score": 0.578, "apex-agents/Overall Pass@1": 0.23, "apex-agents/Overall Pass@8": 0.4, "apex-agents/Overall Mean Score": 0.387, "apex-agents/Investment Banking Pass@1": 0.273, "apex-agents/Management Consulting Pass@1": 0.227, "apex-agents/Corporate Law Pass@1": 0.189, - "apex-agents/Corporate Lawyer Mean Score": 0.443 + "apex-agents/Corporate Lawyer Mean Score": 0.443, + "ace/Overall Score": 0.515, + "ace/Food Score": 0.65, + "ace/Gaming Score": 0.578 } }, { @@ -51591,16 +51620,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.6493, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Safety": 0.8619, - "reward-bench/Reasoning": 0.8661, + "reward-bench/Score": 0.8673, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, + "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819 + "reward-bench/Ties": 0.7819, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Reasoning": 0.8661 } }, { @@ -51678,16 +51707,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.5796, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8374, + "reward-bench/Score": 0.8007, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962 + "reward-bench/Ties": 0.6962, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Reasoning": 0.8374 } }, { @@ -51741,7 +51770,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 43.4 + "terminal-bench-2.0/terminal-bench-2.0": 41.3 } }, { @@ -51750,7 +51779,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 31.9 + "terminal-bench-2.0/terminal-bench-2.0": 22.2 } }, { @@ -51773,7 +51802,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 + "terminal-bench-2.0/terminal-bench-2.0": 7.9 } }, { @@ -51805,7 +51834,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 53.5 + "terminal-bench-2.0/terminal-bench-2.0": 36.9 } }, { @@ -51832,7 +51861,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.7 + "terminal-bench-2.0/terminal-bench-2.0": 54.0 } }, { @@ -51842,13 +51871,13 @@ "evaluator_relationship": null, "benchmark_scores": { "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.26, + "browsecompplus/browsecompplus": 0.48, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.57, - "tau-bench-2_airline/tau-bench-2/airline": 0.6, - "tau-bench-2_retail/tau-bench-2/retail": 0.68, + "swe-bench/swe-bench": 0.5455, + "tau-bench-2_airline/tau-bench-2/airline": 0.48, + "tau-bench-2_retail/tau-bench-2/retail": 0.51, "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, @@ -51867,7 +51896,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.6 + "terminal-bench-2.0/terminal-bench-2.0": 77.3 } }, { @@ -52052,9 +52081,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.0143, - "livecodebenchpro/Medium Problems": 0.2923, - "livecodebenchpro/Easy Problems": 0.8571 + "livecodebenchpro/Hard Problems": 0.014084507042253521, + "livecodebenchpro/Medium Problems": 0.30985915492957744, + "livecodebenchpro/Easy Problems": 0.8873239436619719 } }, { @@ -52198,17 +52227,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.615, + "reward-bench/Score": 0.2653, + "reward-bench/Chat": 0.9246, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.3289, + "reward-bench/Reasoning": 0.5855, + "reward-bench/Prior Sets (0.5 weight)": 0.6801, "reward-bench/Factuality": 0.3979, "reward-bench/Precise IF": 0.2875, "reward-bench/Math": 0.377, - "reward-bench/Safety": 0.5446, "reward-bench/Focus": 0.1535, - "reward-bench/Ties": 0.047, - "reward-bench/Chat": 0.9246, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.5855, - "reward-bench/Prior Sets (0.5 weight)": 0.6801 + "reward-bench/Ties": 0.047 } }, { @@ -52217,17 +52246,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6901, + "reward-bench/Score": 0.2648, + "reward-bench/Chat": 0.8855, + "reward-bench/Chat Hard": 0.4868, + "reward-bench/Safety": 0.3244, + "reward-bench/Reasoning": 0.7752, + "reward-bench/Prior Sets (0.5 weight)": 0.6533, "reward-bench/Factuality": 0.3179, "reward-bench/Precise IF": 0.2625, "reward-bench/Math": 0.3934, - "reward-bench/Safety": 0.6311, "reward-bench/Focus": 0.2707, - "reward-bench/Ties": 0.0198, - "reward-bench/Chat": 0.8855, - "reward-bench/Chat Hard": 0.4868, - "reward-bench/Reasoning": 0.7752, - "reward-bench/Prior Sets (0.5 weight)": 0.6533 + "reward-bench/Ties": 0.0198 } }, { @@ -52250,17 +52279,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6126, + "reward-bench/Score": 0.32, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Safety": 0.3667, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, - "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836 + "reward-bench/Ties": 0.12 } }, { @@ -52283,17 +52312,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8159, + "reward-bench/Score": 0.5806, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Safety": 0.6267, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, - "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172 + "reward-bench/Ties": 0.5972 } }, { @@ -52330,17 +52359,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6903, + "reward-bench/Score": 0.4683, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Safety": 0.5089, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, - "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294 + "reward-bench/Ties": 0.3036 } }, { @@ -53820,17 +53849,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4727, + "reward-bench/Score": 0.1606, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Safety": 0.1422, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, - "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993 + "reward-bench/Ties": -0.01 } }, { @@ -54143,11 +54172,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.276, + "hfopenllm_v2/BBH": 0.274, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3339, - "hfopenllm_v2/MMLU-PRO": 0.1123 + "hfopenllm_v2/GPQA": 0.25, + "hfopenllm_v2/MUSR": 0.3753, + "hfopenllm_v2/MMLU-PRO": 0.112 } }, { @@ -54954,12 +54983,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6064, - "hfopenllm_v2/BBH": 0.6296, - "hfopenllm_v2/MATH Level 5": 0.3708, - "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4873, - "hfopenllm_v2/MMLU-PRO": 0.5307 + "hfopenllm_v2/IFEval": 0.6052, + "hfopenllm_v2/BBH": 0.6317, + "hfopenllm_v2/MATH Level 5": 0.4789, + "hfopenllm_v2/GPQA": 0.3742, + "hfopenllm_v2/MUSR": 0.486, + "hfopenllm_v2/MMLU-PRO": 0.5302 } }, { @@ -56562,12 +56591,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2401, - "hfopenllm_v2/BBH": 0.4622, - "hfopenllm_v2/MATH Level 5": 0.0725, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3703, - "hfopenllm_v2/MMLU-PRO": 0.2379 + "hfopenllm_v2/IFEval": 0.2358, + "hfopenllm_v2/BBH": 0.4612, + "hfopenllm_v2/MATH Level 5": 0.0642, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3717, + "hfopenllm_v2/MMLU-PRO": 0.2382 } }, { @@ -56576,12 +56605,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6066, - "hfopenllm_v2/BBH": 0.635, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.3725, + "hfopenllm_v2/IFEval": 0.6005, + "hfopenllm_v2/BBH": 0.6356, + "hfopenllm_v2/MATH Level 5": 0.2764, + "hfopenllm_v2/GPQA": 0.3691, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5331 + "hfopenllm_v2/MMLU-PRO": 0.5339 } }, { @@ -57514,12 +57543,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7016, - "hfopenllm_v2/BBH": 0.6942, - "hfopenllm_v2/MATH Level 5": 0.4116, - "hfopenllm_v2/GPQA": 0.3624, - "hfopenllm_v2/MUSR": 0.4571, - "hfopenllm_v2/MMLU-PRO": 0.5411 + "hfopenllm_v2/IFEval": 0.2922, + "hfopenllm_v2/BBH": 0.6559, + "hfopenllm_v2/MATH Level 5": 0.2545, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3929, + "hfopenllm_v2/MMLU-PRO": 0.5207 } }, { @@ -59020,12 +59049,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6101, - "hfopenllm_v2/BBH": 0.5008, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3352 + "hfopenllm_v2/IFEval": 0.6147, + "hfopenllm_v2/BBH": 0.4999, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { @@ -59371,17 +59400,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.589, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209, + "reward-bench/Score": 0.8464, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743 + "reward-bench/Ties": 0.5743, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209 } }, { @@ -59408,17 +59437,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6089, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Safety": 0.7867, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309, + "reward-bench/Score": 0.8542, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, + "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981 + "reward-bench/Ties": 0.5981, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309 } }, { @@ -60129,12 +60158,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2595, - "hfopenllm_v2/BBH": 0.3884, - "hfopenllm_v2/MATH Level 5": 0.0914, - "hfopenllm_v2/GPQA": 0.2743, + "hfopenllm_v2/IFEval": 0.2566, + "hfopenllm_v2/BBH": 0.39, + "hfopenllm_v2/MATH Level 5": 0.1208, + "hfopenllm_v2/GPQA": 0.2626, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2719 + "hfopenllm_v2/MMLU-PRO": 0.2741 } }, { @@ -62463,17 +62492,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6292, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492, + "reward-bench/Score": 0.8338, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647 + "reward-bench/Ties": 0.6647, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492 } }, { @@ -62552,16 +62581,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7249, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.9903, + "reward-bench/Score": 0.9499, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, + "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633 + "reward-bench/Ties": 0.7633, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Reasoning": 0.9903 } }, { @@ -63278,16 +63307,16 @@ "hfopenllm_v2/GPQA": 0.344, "hfopenllm_v2/MUSR": 0.4231, "hfopenllm_v2/MMLU-PRO": 0.4103, - "reward-bench/Score": 0.9426, + "reward-bench/Score": 0.7531, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.8991, + "reward-bench/Safety": 0.9689, + "reward-bench/Reasoning": 0.9807, "reward-bench/Factuality": 0.7674, "reward-bench/Precise IF": 0.375, "reward-bench/Math": 0.6721, - "reward-bench/Safety": 0.9297, "reward-bench/Focus": 0.9172, - "reward-bench/Ties": 0.8182, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.8991, - "reward-bench/Reasoning": 0.9807 + "reward-bench/Ties": 0.8182 } }, { @@ -63296,16 +63325,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9252, + "reward-bench/Score": 0.7314, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8728, + "reward-bench/Safety": 0.9333, + "reward-bench/Reasoning": 0.962, "reward-bench/Factuality": 0.6989, "reward-bench/Precise IF": 0.425, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.9616, - "reward-bench/Ties": 0.741, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8728, - "reward-bench/Reasoning": 0.962 + "reward-bench/Ties": 0.741 } }, { @@ -66755,12 +66784,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1829, - "hfopenllm_v2/BBH": 0.5488, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.4056, - "hfopenllm_v2/MMLU-PRO": 0.3805 + "hfopenllm_v2/IFEval": 0.4501, + "hfopenllm_v2/BBH": 0.5472, + "hfopenllm_v2/MATH Level 5": 0.0944, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4017, + "hfopenllm_v2/MMLU-PRO": 0.3792 } }, { @@ -70971,12 +71000,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3496, - "hfopenllm_v2/BBH": 0.4947, - "hfopenllm_v2/MATH Level 5": 0.1269, - "hfopenllm_v2/GPQA": 0.3037, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.3644 + "hfopenllm_v2/IFEval": 0.7168, + "hfopenllm_v2/BBH": 0.4911, + "hfopenllm_v2/MATH Level 5": 0.1533, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.3512, + "hfopenllm_v2/MMLU-PRO": 0.3663 } }, { @@ -71027,12 +71056,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2678, - "hfopenllm_v2/BBH": 0.4429, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.302, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.2927 + "hfopenllm_v2/IFEval": 0.6496, + "hfopenllm_v2/BBH": 0.4774, + "hfopenllm_v2/MATH Level 5": 0.0566, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.3909, + "hfopenllm_v2/MMLU-PRO": 0.3382 } }, { @@ -72365,7 +72394,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.1 + "terminal-bench-2.0/terminal-bench-2.0": 25.4 } }, { @@ -73018,12 +73047,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.177, - "hfopenllm_v2/BBH": 0.3411, - "hfopenllm_v2/MATH Level 5": 0.031, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.374, - "hfopenllm_v2/MMLU-PRO": 0.2529 + "hfopenllm_v2/IFEval": 0.1856, + "hfopenllm_v2/BBH": 0.4149, + "hfopenllm_v2/MATH Level 5": 0.0234, + "hfopenllm_v2/GPQA": 0.276, + "hfopenllm_v2/MUSR": 0.3765, + "hfopenllm_v2/MMLU-PRO": 0.2573 } }, { @@ -73111,12 +73140,12 @@ "developer": "ycros", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5994, - "hfopenllm_v2/BBH": 0.5159, - "hfopenllm_v2/MATH Level 5": 0.0785, - "hfopenllm_v2/GPQA": 0.3045, - "hfopenllm_v2/MUSR": 0.4203, - "hfopenllm_v2/MMLU-PRO": 0.3473 + "hfopenllm_v2/IFEval": 0.6262, + "hfopenllm_v2/BBH": 0.5142, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.3079, + "hfopenllm_v2/MUSR": 0.4138, + "hfopenllm_v2/MMLU-PRO": 0.3481 } }, { @@ -75520,7 +75549,7 @@ "developer": "Z-AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 33.4 + "terminal-bench-2.0/terminal-bench-2.0": 33.3 } }, { diff --git a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json index a8a69697e53329e56da6b7741229b84b4f2eeb85..776a7c662b5afb276bf059bb90896cee5cd968ec 100644 --- a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json +++ b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json @@ -5,7 +5,7 @@ "developer": "adriszmar", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1746 + "score": 0.1685 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3126 + "score": 0.3124 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0015 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.245 + "score": 0.2492 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4096 + "score": 0.3963 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1087 + "score": 0.1066 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1685 + "score": 0.1746 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3124 + "score": 0.3126 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0015 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2492 + "score": 0.245 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3963 + "score": 0.4096 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1066 + "score": 0.1087 } } ], diff --git a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json index 036d2144a17c2773cc45b3fcdc429ec4c85d29b1..e1ac41e4bb97f41e9784803c588108ddfee89905 100644 --- a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json +++ b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6905 + "score": 0.6895 }, "source_data": { "dataset_name": "RewardBench", @@ -152,7 +152,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -170,7 +170,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -188,7 +188,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7676 + "score": 0.7595 }, "source_data": { "dataset_name": "RewardBench", @@ -230,7 +230,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7004 + "score": 0.7019 }, "source_data": { "dataset_name": "RewardBench", @@ -248,7 +248,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9413 + "score": 0.9497 }, "source_data": { "dataset_name": "RewardBench", @@ -266,7 +266,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.375 }, "source_data": { "dataset_name": "RewardBench", @@ -284,7 +284,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7716 + "score": 0.7811 }, "source_data": { "dataset_name": "RewardBench", @@ -326,7 +326,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6945 + "score": 0.6905 }, "source_data": { "dataset_name": "RewardBench", @@ -344,7 +344,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -362,7 +362,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -380,7 +380,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7743 + "score": 0.7676 }, "source_data": { "dataset_name": "RewardBench", @@ -518,7 +518,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7008 + "score": 0.6924 }, "source_data": { "dataset_name": "RewardBench", @@ -536,7 +536,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -554,7 +554,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3575 }, "source_data": { "dataset_name": "RewardBench", @@ -614,7 +614,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7019 + "score": 0.6945 }, "source_data": { "dataset_name": "RewardBench", @@ -632,7 +632,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -650,7 +650,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -668,7 +668,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7811 + "score": 0.7743 }, "source_data": { "dataset_name": "RewardBench", @@ -710,7 +710,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6924 + "score": 0.7004 }, "source_data": { "dataset_name": "RewardBench", @@ -728,7 +728,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9413 }, "source_data": { "dataset_name": "RewardBench", @@ -746,7 +746,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3575 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -764,7 +764,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7716 }, "source_data": { "dataset_name": "RewardBench", @@ -806,7 +806,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6895 + "score": 0.7008 }, "source_data": { "dataset_name": "RewardBench", @@ -842,7 +842,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -860,7 +860,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7595 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", diff --git a/data/models/alibaba_qwen-3-coder-480b.json b/data/models/alibaba_qwen-3-coder-480b.json index 5529ecfec8fec832ba0fc670045bbad34374bc66..f598d15a400b5ed438e43880c6cab3a3248e3641 100644 --- a/data/models/alibaba_qwen-3-coder-480b.json +++ b/data/models/alibaba_qwen-3-coder-480b.json @@ -4,13 +4,13 @@ "id": "alibaba/qwen-3-coder-480b", "developer": "Alibaba", "additional_details": { - "agent_name": "Dakou Agent", - "agent_organization": "iflow" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-28", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 25.4, "uncertainty": { "standard_error": { "value": 2.6 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-12-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 23.9, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.4, + "score": 23.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json b/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json index 3edb4e861c9c885f1d8b857f90568f09968d9fbc..6b82bed55df1bdb62067f41e1f2f588c7fd5e772 100644 --- a/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json +++ b/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9021 + "score": 0.7606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.8126 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8355 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6995 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9095 + "score": 0.8844 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8969 + "score": 0.8646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8835 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.7606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8126 + "score": 0.9021 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.8355 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8844 + "score": 0.9095 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8646 + "score": 0.8969 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8835 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json index 408cdcea425aaae910c976c1f57587d29fcf9c63..52eed39924ec31b815aba5be343333e3d61411ae 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8431 + "score": 0.687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.7516 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.86 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7898 + "score": 0.8545 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.6397 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.687 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7516 + "score": 0.8431 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.86 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8545 + "score": 0.7898 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6397 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b-sft-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-8b-sft-rm-rb2.json index 84c4146c17907f25ce74864c179dc0e7dcd19fbd..e5b15dac5cb2f4285962219bbf6cd6008c5fe41a 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b-sft-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b-sft-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6821 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7326 + "score": 0.8551 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5792 + "score": 0.7917 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8978 + "score": 0.8784 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8889 + "score": 0.8005 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6063 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8551 + "score": 0.6821 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.7326 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7917 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5792 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8784 + "score": 0.8978 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8005 + "score": 0.8889 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.6063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b.json b/data/models/allenai_llama-3.1-tulu-3-8b.json index 7de1c9431728784c04f1c32781631febc2ed7e32..53350f20b1f74556a0af6a9aa87ed77d181deaec 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8255 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.405 + "score": 0.4061 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1964 + "score": 0.2115 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2987 + "score": 0.297 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2821 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8255 + "score": 0.8267 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4061 + "score": 0.405 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2115 + "score": 0.1964 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.297 + "score": 0.2987 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2821 + "score": 0.2827 } } ], diff --git a/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json b/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json new file mode 100644 index 0000000000000000000000000000000000000000..cedfd398cf32e477ed09f75303d207e32a02614e --- /dev/null +++ b/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json @@ -0,0 +1,162 @@ +{ + "model_info": { + "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", + "developer": "allenai", + "additional_details": { + "model_type": "Seq. Classifier" + } + }, + "evaluations": [ + { + "evaluation_id": "reward-bench-2/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ + { + "evaluation_name": "Score", + "metric_config": { + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5151 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Factuality", + "metric_config": { + "evaluation_description": "Factuality score - measures factual accuracy", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6484 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Precise IF", + "metric_config": { + "evaluation_description": "Precise Instruction Following score", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3312 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5574 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Safety", + "metric_config": { + "evaluation_description": "Safety score - measures safety awareness", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.7289 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Focus", + "metric_config": { + "evaluation_description": "Focus score - measures response focus", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.4889 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Ties", + "metric_config": { + "evaluation_description": "Ties score - ability to identify tie cases", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3357 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + } + ] +} \ No newline at end of file diff --git a/data/models/amd_amd-llama-135m.json b/data/models/amd_amd-llama-135m.json index dfdcf8e7a431ddc8bdf51b6aa797b81cf0f89c9e..a445333e3dd0495a4c3e2fddf3e92a96b288f85c 100644 --- a/data/models/amd_amd-llama-135m.json +++ b/data/models/amd_amd-llama-135m.json @@ -5,9 +5,9 @@ "developer": "amd", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", - "params_billions": "0.134" + "params_billions": "0.135" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1918 + "score": 0.1842 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2969 + "score": 0.2974 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0053 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2584 + "score": 0.2525 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3846 + "score": 0.378 } }, { @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1842 + "score": 0.1918 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2974 + "score": 0.2969 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0053 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2525 + "score": 0.2584 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.378 + "score": 0.3846 } }, { diff --git a/data/models/anthropic_claude-3-5-haiku-20241022.json b/data/models/anthropic_claude-3-5-haiku-20241022.json index 9edb0c7484cd3b1eab61fa90716227281f955253..97ab32cadbb5e110dfa409cbea1ee9b8c38d6850 100644 --- a/data/models/anthropic_claude-3-5-haiku-20241022.json +++ b/data/models/anthropic_claude-3-5-haiku-20241022.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-5-sonnet-20240620.json b/data/models/anthropic_claude-3-5-sonnet-20240620.json index db73167f2ee9a0684ae3211521c8476cfd4cde1f..3864d6fb81d70dd8f8c2977e067122577576cb3c 100644 --- a/data/models/anthropic_claude-3-5-sonnet-20240620.json +++ b/data/models/anthropic_claude-3-5-sonnet-20240620.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-5-sonnet-20240620/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-5-sonnet-20240620/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,128 +1925,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8417 + "score": 0.6466 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.5284 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7401 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.5683 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8469 + "score": 0.8519 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/anthropic_claude-3-5-sonnet-20240620/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6466 + "score": 0.8697 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2055,111 +2031,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5284 + "score": 0.674 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Anthropic_claude-3-5-sonnet-20240620/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.8417 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5683 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8519 + "score": 0.7401 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8697 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.674 + "score": 0.8469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/anthropic_claude-3-haiku-20240307.json b/data/models/anthropic_claude-3-haiku-20240307.json index 4a2b563f634d87e24d77a5003bd4befff9181e63..52bb9959dd5613be4f3fe143bd6f01f97e2632c2 100644 --- a/data/models/anthropic_claude-3-haiku-20240307.json +++ b/data/models/anthropic_claude-3-haiku-20240307.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,127 +1925,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3711 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4042 + "score": 0.7289 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.9274 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3552 + "score": 0.5197 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.595 + "score": 0.7953 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.501 + "score": 0.706 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0899 + "score": 0.6635 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -2053,10 +2035,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2075,109 +2057,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7289 + "score": 0.3711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9274 + "score": 0.4042 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5197 + "score": 0.2812 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3552 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7953 + "score": 0.595 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.706 + "score": 0.501 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6635 + "score": 0.0899 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/anthropic_claude-3-opus-20240229.json b/data/models/anthropic_claude-3-opus-20240229.json index 150503f1a5f91571d10a33f8965d4bd84f3ccf46..8f8b70dd804fcbb225ceb8111209d93e20e8bd0b 100644 --- a/data/models/anthropic_claude-3-opus-20240229.json +++ b/data/models/anthropic_claude-3-opus-20240229.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", + "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,104 +1925,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5744 + "score": 0.8008 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5389 + "score": 0.9469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.6031 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5137 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8378 + "score": 0.7868 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6646 + "score": 0.5744 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2031,135 +2055,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5601 + "score": 0.5389 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8008 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9469 + "score": 0.5137 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.8378 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.6646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7868 + "score": 0.5601 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/anthropic_claude-haiku-4.5.json b/data/models/anthropic_claude-haiku-4.5.json index a44b80f840579ff8c5f06292af8a76c394e19084..0030689870b1d61c70e99ee0cdff0afc1d0e9bf2 100644 --- a/data/models/anthropic_claude-haiku-4.5.json +++ b/data/models/anthropic_claude-haiku-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-haiku-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.8, + "score": 28.3, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.5, + "score": 27.5, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 13.9, + "score": 29.8, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 28.3, + "score": 13.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.5, + "score": 35.5, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4-1-20250805.json b/data/models/anthropic_claude-opus-4-1-20250805.json index d8d76da387da094a25675bc9e4c49b25e5dc839b..97af67f7a9590f836aeb06cb6e8ac5e127131091 100644 --- a/data/models/anthropic_claude-opus-4-1-20250805.json +++ b/data/models/anthropic_claude-opus-4-1-20250805.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-opus-4-5.json b/data/models/anthropic_claude-opus-4-5.json index e58ae243e6675ffbe5718b2cd53dbb80e8f1ba8b..0cd69b8f1e682224797162bd821dc65c4792b021 100644 --- a/data/models/anthropic_claude-opus-4-5.json +++ b/data/models/anthropic_claude-opus-4-5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4-5", "developer": "Anthropic", "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -23,42 +23,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.49, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "22.76", - "total_run_cost": "2276.48", - "average_steps": "47.65", - "percent_finished": "0.77" + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.61, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "13.08", - "total_run_cost": "1308.38", - "average_steps": "49.69", - "percent_finished": "0.74" + "average_agent_cost": "11.32", + "total_run_cost": "1132.47", + "average_steps": "21.99", + "percent_finished": "0.83" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -159,42 +159,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.5294, "uncertainty": { - "num_samples": 100 + "num_samples": 51 }, "details": { - "average_agent_cost": "11.32", - "total_run_cost": "1132.47", - "average_steps": "21.99", - "percent_finished": "0.83" + "average_agent_cost": "11.66", + "total_run_cost": "594.68", + "average_steps": "31.04", + "percent_finished": "0.8431" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -206,8 +206,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -282,7 +282,7 @@ } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.66, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "5.59", - "total_run_cost": "558.51", - "average_steps": "41.07", - "percent_finished": "0.82" + "average_agent_cost": "13.08", + "total_run_cost": "1308.38", + "average_steps": "49.69", + "percent_finished": "0.74" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,23 +382,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.64, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.43", - "total_run_cost": "343.32", - "average_steps": "20.06", - "percent_finished": "0.82" + "average_agent_cost": "22.76", + "total_run_cost": "2276.48", + "average_steps": "47.65", + "percent_finished": "0.77" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5294, + "score": 0.61, "uncertainty": { - "num_samples": 51 + "num_samples": 100 }, "details": { - "average_agent_cost": "11.66", - "total_run_cost": "594.68", - "average_steps": "31.04", - "percent_finished": "0.8431" + "average_agent_cost": "7.59", + "total_run_cost": "759.44", + "average_steps": "27.18", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -499,42 +499,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.7, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "5.59", + "total_run_cost": "558.51", + "average_steps": "41.07", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -601,8 +601,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -635,42 +635,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.64, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.59", - "total_run_cost": "759.44", - "average_steps": "27.18", - "percent_finished": "1.0" + "average_agent_cost": "3.43", + "total_run_cost": "343.32", + "average_steps": "20.06", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -682,15 +682,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -722,14 +722,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.65, + "score": 0.6061, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "4.85", - "total_run_cost": "485.22", - "average_steps": "39.13", + "average_agent_cost": "3.97", + "total_run_cost": "393.16", + "average_steps": "43.44", "percent_finished": "1.0" } }, @@ -737,8 +737,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -750,15 +750,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -790,14 +790,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8072, + "score": 0.65, "uncertainty": { - "num_samples": 83 + "num_samples": 100 }, "details": { - "average_agent_cost": "2.96", - "total_run_cost": "245.78", - "average_steps": "34.1", + "average_agent_cost": "4.85", + "total_run_cost": "485.22", + "average_steps": "39.13", "percent_finished": "1.0" } }, @@ -805,8 +805,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -818,15 +818,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -858,14 +858,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7423, + "score": 0.6061, "uncertainty": { - "num_samples": 97 + "num_samples": 99 }, "details": { - "average_agent_cost": "5.6", - "total_run_cost": "543.62", - "average_steps": "31.76", + "average_agent_cost": "3.97", + "total_run_cost": "393.16", + "average_steps": "43.44", "percent_finished": "1.0" } }, @@ -873,8 +873,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -886,15 +886,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -926,14 +926,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6061, + "score": 0.7423, "uncertainty": { - "num_samples": 99 + "num_samples": 97 }, "details": { - "average_agent_cost": "3.97", - "total_run_cost": "393.16", - "average_steps": "43.44", + "average_agent_cost": "5.6", + "total_run_cost": "543.62", + "average_steps": "31.76", "percent_finished": "1.0" } }, @@ -941,8 +941,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -954,15 +954,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -975,33 +975,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6061, + "score": 0.74, "uncertainty": { - "num_samples": 99 + "num_samples": 50 }, "details": { - "average_agent_cost": "3.97", - "total_run_cost": "393.16", - "average_steps": "43.44", + "average_agent_cost": "0.72", + "total_run_cost": "36.55", + "average_steps": "12.22", "percent_finished": "1.0" } }, @@ -1009,8 +1009,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1022,15 +1022,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1067,9 +1067,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "1.3", + "total_run_cost": "65.66", + "average_steps": "11.5", "percent_finished": "1.0" } }, @@ -1077,8 +1077,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1090,15 +1090,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1135,9 +1135,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "65.66", - "average_steps": "11.5", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1145,8 +1145,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1158,15 +1158,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1198,14 +1198,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.74, + "score": 0.72, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.72", - "total_run_cost": "36.55", - "average_steps": "12.22", + "average_agent_cost": "0.78", + "total_run_cost": "39.67", + "average_steps": "11.88", "percent_finished": "1.0" } }, @@ -1213,8 +1213,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1226,15 +1226,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1247,33 +1247,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.8072, "uncertainty": { - "num_samples": 50 + "num_samples": 83 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "2.96", + "total_run_cost": "245.78", + "average_steps": "34.1", "percent_finished": "1.0" } }, @@ -1281,8 +1281,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1294,15 +1294,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1334,14 +1334,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.72, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.78", - "total_run_cost": "39.67", - "average_steps": "11.88", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1349,8 +1349,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1362,8 +1362,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1438,7 +1438,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1470,14 +1470,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.83, + "score": 0.78, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.6", - "total_run_cost": "161.14", - "average_steps": "12.54", + "average_agent_cost": "0.47", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1485,8 +1485,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1498,15 +1498,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1538,14 +1538,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.83, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "1.6", + "total_run_cost": "161.14", + "average_steps": "12.54", "percent_finished": "1.0" } }, @@ -1553,8 +1553,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1566,15 +1566,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1621,8 +1621,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1634,8 +1634,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1710,7 +1710,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1742,14 +1742,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.06", - "total_run_cost": "114.62", - "average_steps": "13.77", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1757,8 +1757,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1770,15 +1770,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1810,14 +1810,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.45", - "total_run_cost": "255.97", - "average_steps": "18.71", + "average_agent_cost": "1.06", + "total_run_cost": "114.62", + "average_steps": "13.77", "percent_finished": "1.0" } }, @@ -1825,8 +1825,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1838,15 +1838,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1883,9 +1883,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "2.45", + "total_run_cost": "255.97", + "average_steps": "18.71", "percent_finished": "1.0" } }, @@ -1893,8 +1893,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1906,15 +1906,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1946,14 +1946,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.84, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.25", - "total_run_cost": "136.84", - "average_steps": "17.15", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1961,8 +1961,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1974,15 +1974,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2014,14 +2014,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.84, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "1.25", + "total_run_cost": "136.84", + "average_steps": "17.15", "percent_finished": "1.0" } }, @@ -2029,8 +2029,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2042,8 +2042,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } diff --git a/data/models/anthropic_claude-opus-4.1.json b/data/models/anthropic_claude-opus-4.1.json index 59416d90b1dbdb120915f9f1b242f981bce8e890..66d9a9691d54a3cae59a5a4b0972ad6110b06a9b 100644 --- a/data/models/anthropic_claude-opus-4.1.json +++ b/data/models/anthropic_claude-opus-4.1.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.1", "developer": "Anthropic", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 35.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 38.0, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.1, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 38.0, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.5.json b/data/models/anthropic_claude-opus-4.5.json index 007d3ac2c88dec5dff72fba0205d56eeaed42464..08f15dd6e988a559ca38856b6df4c4e4d53b00a1 100644 --- a/data/models/anthropic_claude-opus-4.5.json +++ b/data/models/anthropic_claude-opus-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "OpenCode", + "agent_organization": "Anomaly Innovations" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-22", + "evaluation_timestamp": "2026-01-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,11 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 51.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +64,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -152,7 +146,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -176,7 +170,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-12", + "evaluation_timestamp": "2025-11-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -185,11 +179,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7 + "score": 57.8, + "uncertainty": { + "standard_error": { + "value": 2.5 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -206,7 +206,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -220,7 +220,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -244,7 +244,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -253,17 +253,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.3, + "score": 51.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -280,7 +280,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -294,7 +294,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -318,7 +318,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -327,17 +327,17 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 54.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -354,7 +354,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -368,7 +368,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -392,7 +392,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -401,17 +401,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.9, + "score": 63.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -428,7 +428,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -442,7 +442,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -466,7 +466,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -475,17 +475,17 @@ "max_score": 100.0 }, "score_details": { - "score": 63.1, + "score": 52.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -502,7 +502,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -516,7 +516,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -540,7 +540,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -549,17 +549,17 @@ "max_score": 100.0 }, "score_details": { - "score": 52.1, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -576,7 +576,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.6.json b/data/models/anthropic_claude-opus-4.6.json index 6de13fc0f3140d257d348a852063485bd1007995..1ae6eba958aa9144fbbc47e84b576827e9ba8012 100644 --- a/data/models/anthropic_claude-opus-4.6.json +++ b/data/models/anthropic_claude-opus-4.6.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.6", "developer": "Anthropic", "additional_details": { - "agent_name": "Claude Code", - "agent_organization": "Anthropic" + "agent_name": "TongAgents", + "agent_organization": "Bigai" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-07", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.0, + "score": 71.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-13", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 66.5, + "score": 69.9, "uncertainty": { "standard_error": { "value": 2.5 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.9, + "score": 74.7, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-02-13", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,11 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.9 + "score": 66.5, + "uncertainty": { + "standard_error": { + "value": 2.5 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -398,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -407,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 69.9, + "score": 58.0, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +454,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +478,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +487,11 @@ "max_score": 100.0 }, "score_details": { - "score": 74.7, - "uncertainty": { - "standard_error": { - "value": 2.6 - }, - "num_samples": 435 - } + "score": 66.9 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-sonnet-4-20250514.json b/data/models/anthropic_claude-sonnet-4-20250514.json index a43572d10d77d034b8cc0b4e9e80cb595c19f907..17485295f0940239338e7d9ae5edbc1db77fa9eb 100644 --- a/data/models/anthropic_claude-sonnet-4-20250514.json +++ b/data/models/anthropic_claude-sonnet-4-20250514.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-sonnet-4.5.json b/data/models/anthropic_claude-sonnet-4.5.json index 66356dbecae8af8177ad511141877805ad8752da..c0439a05bb694a022b80d498957deb384990a100 100644 --- a/data/models/anthropic_claude-sonnet-4.5.json +++ b/data/models/anthropic_claude-sonnet-4.5.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.8, + "score": 46.5, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,11 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.7 + "score": 42.5, + "uncertainty": { + "standard_error": { + "value": 2.8 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 42.6, + "score": 42.8, "uncertainty": { "standard_error": { "value": 2.8 @@ -269,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 40.1, + "score": 42.6, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -398,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -407,17 +413,11 @@ "max_score": 100.0 }, "score_details": { - "score": 46.5, - "uncertainty": { - "standard_error": { - "value": 2.4 - }, - "num_samples": 435 - } + "score": 42.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.5, + "score": 40.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json b/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json index d9c4a63a2048d58930bc10ecb5695a3676976b8f..c02cc4e6043653215ef5039a1e40518d83ff39e0 100644 --- a/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json +++ b/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json @@ -5,7 +5,7 @@ "developer": "AtAndDev", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "1.544" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4511 + "score": 0.4605 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4275 + "score": 0.4258 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1473 + "score": 0.0748 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3623 + "score": 0.3636 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2806 + "score": 0.2812 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4605 + "score": 0.4511 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4258 + "score": 0.4275 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0748 + "score": 0.1473 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3636 + "score": 0.3623 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.2806 } } ], diff --git a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json index ed5c0ad9e23e3e1024094c595c2931cbc579e102..eb03ad8b35e71a4f26603a2960132d29f0246b60 100644 --- a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json +++ b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json @@ -5,7 +5,7 @@ "developer": "BoltMonkey", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.459 + "score": 0.7999 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5185 + "score": 0.5152 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.1193 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.281 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4083 + "score": 0.4019 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3631 + "score": 0.3733 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7999 + "score": 0.459 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5152 + "score": 0.5185 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1193 + "score": 0.0937 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.281 + "score": 0.2743 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.4083 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3631 } } ], diff --git a/data/models/bunnycore_llama-3.2-3b-deep-test.json b/data/models/bunnycore_llama-3.2-3b-deep-test.json index c829321bc4e1b62229f276602c62e6f6bd9a3faf..05cbb770cffa010b799551a630bc7b1089a91900 100644 --- a/data/models/bunnycore_llama-3.2-3b-deep-test.json +++ b/data/models/bunnycore_llama-3.2-3b-deep-test.json @@ -5,9 +5,9 @@ "developer": "bunnycore", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", - "params_billions": "1.803" + "params_billions": "3.607" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1775 + "score": 0.4652 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.295 + "score": 0.4531 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1284 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2643 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3647 + "score": 0.3394 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1049 + "score": 0.3152 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4652 + "score": 0.1775 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.295 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1284 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2643 + "score": 0.2517 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3394 + "score": 0.3647 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3152 + "score": 0.1049 } } ], diff --git a/data/models/cir-ams_btrm_qwen2_7b_0613.json b/data/models/cir-ams_btrm_qwen2_7b_0613.json index 84a36ab2262aa5d4869e5d141d383b85699971f1..9fb827ce32fb32044e2247d7f86c70d1bc13d414 100644 --- a/data/models/cir-ams_btrm_qwen2_7b_0613.json +++ b/data/models/cir-ams_btrm_qwen2_7b_0613.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8172 + "score": 0.5736 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9749 + "score": 0.5347 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3563 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6066 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9014 + "score": 0.7178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8775 + "score": 0.5737 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7029 + "score": 0.6527 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5736 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.8172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3563 + "score": 0.9749 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7178 + "score": 0.9014 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5737 + "score": 0.8775 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6527 + "score": 0.7029 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/cohere_command-a-03-2025.json b/data/models/cohere_command-a-03-2025.json index aedaf4e5f944a400d87afc543fa422bde24be135..205f7c496ed81b612dd2b182b3cf2e9fd9ac9c54 100644 --- a/data/models/cohere_command-a-03-2025.json +++ b/data/models/cohere_command-a-03-2025.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/cpayne1303_llama-43m-beta.json b/data/models/cpayne1303_llama-43m-beta.json index 8e62d1be9edc1ccaeda7702d1a793fd5db4d4150..0f0d3430f35b22a1aef8ac071591532d594d6d4d 100644 --- a/data/models/cpayne1303_llama-43m-beta.json +++ b/data/models/cpayne1303_llama-43m-beta.json @@ -5,7 +5,7 @@ "developer": "cpayne1303", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "0.043" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1916 + "score": 0.1949 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2965 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0045 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3872 + "score": 0.3885 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1132 + "score": 0.1111 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1949 + "score": 0.1916 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2965 + "score": 0.2977 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0045 + "score": 0.0 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3885 + "score": 0.3872 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1111 + "score": 0.1132 } } ], diff --git a/data/models/daemontatox_pathfinderai.json b/data/models/daemontatox_pathfinderai.json index 7a5f7d25c7278e2df08548a48abfe0b0ee9b4f2a..8b13e2aabf79ff36b82f8d8bd4c0bcdf2b41e385 100644 --- a/data/models/daemontatox_pathfinderai.json +++ b/data/models/daemontatox_pathfinderai.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4855 + "score": 0.3745 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6627 + "score": 0.6668 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4841 + "score": 0.4758 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3943 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4256 + "score": 0.4858 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5542 + "score": 0.5593 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3745 + "score": 0.4855 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6668 + "score": 0.6627 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4841 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.3096 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4858 + "score": 0.4256 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5593 + "score": 0.5542 } } ], diff --git a/data/models/davielion_llama-3.2-1b-spin-iter0.json b/data/models/davielion_llama-3.2-1b-spin-iter0.json index 2ea14b6e6da41d61b06880761473f45de63d9756..849a1d52d59a5f16eab6aaa35f259f630d9dc175 100644 --- a/data/models/davielion_llama-3.2-1b-spin-iter0.json +++ b/data/models/davielion_llama-3.2-1b-spin-iter0.json @@ -5,7 +5,7 @@ "developer": "DavieLion", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1507 + "score": 0.1549 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.293 + "score": 0.2937 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.006 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2576 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1125 + "score": 0.1128 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1549 + "score": 0.1507 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2937 + "score": 0.293 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.006 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2534 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1125 } } ], diff --git a/data/models/dfurman_llama-3-8b-orpo-v0.1.json b/data/models/dfurman_llama-3-8b-orpo-v0.1.json index 987b3112f03b34626e0421787b1c9cb2e2b55a46..a2dcf54fc8dfd61d5182710e882025f5ed46cdab 100644 --- a/data/models/dfurman_llama-3-8b-orpo-v0.1.json +++ b/data/models/dfurman_llama-3-8b-orpo-v0.1.json @@ -5,8 +5,8 @@ "developer": "dfurman", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", - "architecture": "LlamaForCausalLM", + "precision": "float16", + "architecture": "?", "params_billions": "8.03" } }, @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3 + "score": 0.2835 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3853 + "score": 0.3842 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0415 + "score": 0.0521 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2617 + "score": 0.2609 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3579 + "score": 0.3566 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2281 + "score": 0.2298 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2835 + "score": 0.3 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.3853 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0415 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2617 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3566 + "score": 0.3579 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2298 + "score": 0.2281 } } ], diff --git a/data/models/doppelreflex_mn-12b-lilithframe.json b/data/models/doppelreflex_mn-12b-lilithframe.json index 720fbe0846130d5605f8d9f0e7d4c1731ec8e47c..2532c65fe82e1d06f8d68b438dbffc94e3e4e4bb 100644 --- a/data/models/doppelreflex_mn-12b-lilithframe.json +++ b/data/models/doppelreflex_mn-12b-lilithframe.json @@ -5,7 +5,7 @@ "developer": "DoppelReflEx", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "12.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.451 + "score": 0.436 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4944 + "score": 0.4956 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1156 + "score": 0.0589 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3196 + "score": 0.3205 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3896 + "score": 0.3843 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3256 + "score": 0.3237 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.436 + "score": 0.451 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4956 + "score": 0.4944 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0589 + "score": 0.1156 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3205 + "score": 0.3196 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3843 + "score": 0.3896 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3237 + "score": 0.3256 } } ], diff --git a/data/models/google_flan-t5-xl.json b/data/models/google_flan-t5-xl.json index f4504e45576f272b296c7e91d2af835d208d50a9..55626bf61977dcd421335d97681a7abdf7372f51 100644 --- a/data/models/google_flan-t5-xl.json +++ b/data/models/google_flan-t5-xl.json @@ -5,7 +5,7 @@ "developer": "Google", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "T5ForConditionalGeneration", "params_billions": "2.85" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2237 + "score": 0.2207 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.4537 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0008 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2525 + "score": 0.2458 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4181 + "score": 0.422 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2147 + "score": 0.2142 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2207 + "score": 0.2237 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4537 + "score": 0.4531 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0008 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2458 + "score": 0.2525 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.422 + "score": 0.4181 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2142 + "score": 0.2147 } } ], diff --git a/data/models/google_gemini-2.5-flash-preview-05-20.json b/data/models/google_gemini-2.5-flash-preview-05-20.json index b4b59bae4ffec304ac091e2d0d21ff7a255dc5af..01b694cbaacb7c97812a9bb056c39eebb70c1f12 100644 --- a/data/models/google_gemini-2.5-flash-preview-05-20.json +++ b/data/models/google_gemini-2.5-flash-preview-05-20.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash-preview-05-20/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash-preview-05-20/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash-preview-05-20/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash-preview-05-20/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/google_gemini-2.5-flash.json b/data/models/google_gemini-2.5-flash.json index 5eaa67bbe68ea1014fddc37d9abaae6354738cf4..e86caee99d92f5b1111f5d4cfe31c4107b3564b0 100644 --- a/data/models/google_gemini-2.5-flash.json +++ b/data/models/google_gemini-2.5-flash.json @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 15.4, + "score": 16.9, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 17.1, + "score": 16.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 15.4, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1491,7 +1491,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1515,7 +1515,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1524,17 +1524,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.9, + "score": 17.1, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1551,7 +1551,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-2.5-pro.json b/data/models/google_gemini-2.5-pro.json index ad69bc6f71351eff92acf181b72bb1ef3d2ca29a..f2d7e76912c502afd2dae0560e9a20a114690132 100644 --- a/data/models/google_gemini-2.5-pro.json +++ b/data/models/google_gemini-2.5-pro.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 32.6, + "score": 26.1, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.1, + "score": 16.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 19.6, + "score": 32.6, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1491,7 +1491,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1515,7 +1515,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1524,17 +1524,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 19.6, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1551,7 +1551,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-flash.json b/data/models/google_gemini-3-flash.json index 6d70bb6cf77fa0c54cb9eb4fa0ff050319c617e5..8ad1fcf1488cbfb100a465260bd9105159bfdda1 100644 --- a/data/models/google_gemini-3-flash.json +++ b/data/models/google_gemini-3-flash.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Junie CLI", + "agent_organization": "JetBrains" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-07", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7, + "score": 64.3, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 47.4, + "score": 51.7, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.3, + "score": 47.4, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-pro-preview.json b/data/models/google_gemini-3-pro-preview.json index fdc61e428a001318c9e1c391ace7515f63b19817..3f20467a04db22c43c2d253a7e0a111a72926152 100644 --- a/data/models/google_gemini-3-pro-preview.json +++ b/data/models/google_gemini-3-pro-preview.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro-preview", "developer": "Google", "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.13, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.54", - "total_run_cost": "254.25", - "average_steps": "49.13", - "percent_finished": "0.71" + "average_agent_cost": "1.3", + "total_run_cost": "130.49", + "average_steps": "22.59", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.505, + "score": 0.582, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.88", - "total_run_cost": "188.19", - "average_steps": "21.76", - "percent_finished": "0.99" + "average_agent_cost": "8.7", + "total_run_cost": "869.55", + "average_steps": "33.49", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.582, + "score": 0.505, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "8.7", - "total_run_cost": "869.55", - "average_steps": "33.49", - "percent_finished": "0.98" + "average_agent_cost": "1.88", + "total_run_cost": "188.19", + "average_steps": "21.76", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -246,23 +246,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.36, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "130.49", - "average_steps": "22.59", - "percent_finished": "1.0" + "average_agent_cost": "3.11", + "total_run_cost": "310.55", + "average_steps": "38.01", + "percent_finished": "0.86" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.3333, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "2.39", - "total_run_cost": "239.0", - "average_steps": "29.63", - "percent_finished": "0.69" + "average_agent_cost": "0.64", + "total_run_cost": "63.79", + "average_steps": "8.45", + "percent_finished": "0.6061" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -363,34 +363,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.36, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.11", - "total_run_cost": "310.55", - "average_steps": "38.01", - "percent_finished": "0.86" + "average_agent_cost": "2.85", + "total_run_cost": "284.68", + "average_steps": "22.88", + "percent_finished": "0.7" } }, "generation_config": { @@ -418,7 +418,7 @@ } }, { - "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -431,42 +431,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.13, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.85", - "total_run_cost": "284.68", - "average_steps": "22.88", - "percent_finished": "0.7" + "average_agent_cost": "2.54", + "total_run_cost": "254.25", + "average_steps": "49.13", + "percent_finished": "0.71" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3333, + "score": 0.48, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.64", - "total_run_cost": "63.79", - "average_steps": "8.45", - "percent_finished": "0.6061" + "average_agent_cost": "0.44", + "total_run_cost": "44.18", + "average_steps": "7.85", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.57, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.44", - "total_run_cost": "44.18", - "average_steps": "7.85", - "percent_finished": "0.99" + "average_agent_cost": "2.39", + "total_run_cost": "239.0", + "average_steps": "29.63", + "percent_finished": "0.69" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1720,7 +1720,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1752,14 +1752,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.67, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.7", - "total_run_cost": "69.56", - "average_steps": "32.55", + "average_agent_cost": "3.68", + "total_run_cost": "367.97", + "average_steps": "43.72", "percent_finished": "1.0" } }, @@ -1767,8 +1767,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1780,8 +1780,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1856,7 +1856,7 @@ } }, { - "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1869,33 +1869,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7234, + "score": 0.62, "uncertainty": { - "num_samples": 94 + "num_samples": 50 }, "details": { - "average_agent_cost": "1.58", - "total_run_cost": "148.44", - "average_steps": "32.36", + "average_agent_cost": "0.21", + "total_run_cost": "11.18", + "average_steps": "10.9", "percent_finished": "1.0" } }, @@ -1924,7 +1924,7 @@ } }, { - "evaluation_id": "swe-bench/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1956,14 +1956,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.67, + "score": 0.71, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.68", - "total_run_cost": "367.97", - "average_steps": "43.72", + "average_agent_cost": "0.7", + "total_run_cost": "69.56", + "average_steps": "32.55", "percent_finished": "1.0" } }, @@ -1971,8 +1971,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1984,15 +1984,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2039,8 +2039,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2052,15 +2052,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2073,33 +2073,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.7234, "uncertainty": { - "num_samples": 50 + "num_samples": 94 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "1.58", + "total_run_cost": "148.44", + "average_steps": "32.36", "percent_finished": "1.0" } }, @@ -2107,8 +2107,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2120,8 +2120,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2196,7 +2196,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2228,14 +2228,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.62, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.18", - "average_steps": "10.9", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2243,8 +2243,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2256,15 +2256,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2296,14 +2296,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.68, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.2", + "total_run_cost": "10.29", + "average_steps": "12.28", "percent_finished": "1.0" } }, @@ -2311,8 +2311,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2324,15 +2324,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2364,14 +2364,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.2", - "total_run_cost": "10.29", - "average_steps": "12.28", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2379,8 +2379,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2392,15 +2392,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2447,8 +2447,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2460,15 +2460,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2500,14 +2500,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "21.43", - "average_steps": "11.3", + "average_agent_cost": "0.27", + "total_run_cost": "27.48", + "average_steps": "10.62", "percent_finished": "1.0" } }, @@ -2515,8 +2515,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2528,15 +2528,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2583,8 +2583,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2596,15 +2596,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2636,14 +2636,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.7576, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.27", - "total_run_cost": "27.48", - "average_steps": "10.62", + "average_agent_cost": "0.21", + "total_run_cost": "21.43", + "average_steps": "11.3", "percent_finished": "1.0" } }, @@ -2651,8 +2651,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2664,15 +2664,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2685,33 +2685,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_telecom", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/telecom", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/telecom", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7805, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.19", - "total_run_cost": "19.38", - "average_steps": "11.18", + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", "percent_finished": "1.0" } }, @@ -2719,8 +2719,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2732,15 +2732,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2772,23 +2772,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6852, + "score": 0.8876, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "25.48", - "average_steps": "9.9", - "percent_finished": "1.0" + "average_agent_cost": "0.54", + "total_run_cost": "58.29", + "average_steps": "10.82", + "percent_finished": "0.89" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2800,15 +2800,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2840,14 +2840,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.88, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.35", + "total_run_cost": "40.25", + "average_steps": "12.71", "percent_finished": "1.0" } }, @@ -2855,8 +2855,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2868,15 +2868,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2889,33 +2889,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_telecom", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/telecom", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/telecom", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.88, + "score": 0.7805, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.35", - "total_run_cost": "40.25", - "average_steps": "12.71", + "average_agent_cost": "0.19", + "total_run_cost": "19.38", + "average_steps": "11.18", "percent_finished": "1.0" } }, @@ -2923,8 +2923,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2936,15 +2936,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2976,23 +2976,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8876, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.54", - "total_run_cost": "58.29", - "average_steps": "10.82", - "percent_finished": "0.89" + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -3004,15 +3004,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -3044,14 +3044,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.6852, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.21", + "total_run_cost": "25.48", + "average_steps": "9.9", "percent_finished": "1.0" } }, @@ -3059,8 +3059,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -3072,8 +3072,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } diff --git a/data/models/google_gemini-3-pro.json b/data/models/google_gemini-3-pro.json index 3f400faaadf810745e49d759105088e642b483cc..5acfd6f3e4c0e364f178f7fde5f99550b7d69ddb 100644 --- a/data/models/google_gemini-3-pro.json +++ b/data/models/google_gemini-3-pro.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro", "developer": "Google", "additional_details": { - "agent_name": "SageAgent", - "agent_organization": "OpenSage" + "agent_name": "Droid", + "agent_organization": "Factory" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/sageagent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 65.2, + "score": 61.1, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/sageagent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-06", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 69.4, + "score": 65.2, "uncertainty": { "standard_error": { "value": 2.1 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-11-21", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.1, + "score": 56.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.2, + "score": 61.8, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.8, + "score": 56.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2026-01-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.0, + "score": 69.4, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -454,7 +454,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -478,7 +478,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-21", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -487,17 +487,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.9, + "score": 62.2, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -514,7 +514,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini_3_flash.json b/data/models/google_gemini_3_flash.json index 3a911939d887cdc9433884fdde5ee11c97c277a4..6beb736773081d366f65edf05d1bf54bd7e07ce2 100644 --- a/data/models/google_gemini_3_flash.json +++ b/data/models/google_gemini_3_flash.json @@ -6,53 +6,6 @@ "inference_platform": "unknown" }, "evaluations": [ - { - "evaluation_id": "ace/google_gemini-3-flash/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.415 - }, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - }, { "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", @@ -252,6 +205,53 @@ } } }, + { + "evaluation_id": "ace/google_gemini-3-flash/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.415 + }, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + }, { "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/google_gemma-2-2b-jpn-it.json b/data/models/google_gemma-2-2b-jpn-it.json index 208075f00df888593e36a118e84f64272df933ff..ad8eb46cb13cde4ab2a8a3521d4c0b618fb71ee6 100644 --- a/data/models/google_gemma-2-2b-jpn-it.json +++ b/data/models/google_gemma-2-2b-jpn-it.json @@ -5,7 +5,7 @@ "developer": "Google", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "2.614" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5078 + "score": 0.5288 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4226 + "score": 0.4178 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0347 + "score": 0.0476 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2852 + "score": 0.2752 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3964 + "score": 0.3728 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2578 + "score": 0.2467 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5288 + "score": 0.5078 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4178 + "score": 0.4226 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0476 + "score": 0.0347 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.2852 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.3964 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2467 + "score": 0.2578 } } ], diff --git a/data/models/google_gemma-3-27b-it.json b/data/models/google_gemma-3-27b-it.json index 31e90c4548397bec1dec70a558a4830dc0c4f7c9..0d22aa7a55f613493f23d93430a44590b7aa715d 100644 --- a/data/models/google_gemma-3-27b-it.json +++ b/data/models/google_gemma-3-27b-it.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/gunulhona_gemma-ko-merge-peft.json b/data/models/gunulhona_gemma-ko-merge-peft.json index 7aaca8e76ac4d54a166ddba85afcaa38d76748c4..632db743d31cc31d893a565c873ea7be7cc73fb2 100644 --- a/data/models/gunulhona_gemma-ko-merge-peft.json +++ b/data/models/gunulhona_gemma-ko-merge-peft.json @@ -5,7 +5,7 @@ "developer": "Gunulhona", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "?", "params_billions": "20.318" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4441 + "score": 0.288 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4863 + "score": 0.5154 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3247 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3986 + "score": 0.408 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3098 + "score": 0.3817 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.288 + "score": 0.4441 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5154 + "score": 0.4863 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.307 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.408 + "score": 0.3986 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3817 + "score": 0.3098 } } ], diff --git a/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json b/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json index 69bdc21b5b14ca68b6163472e8eef6fd89e268f5..357f438b07831c53712dd63f870b6b2401c4d681 100644 --- a/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json +++ b/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7847 + "score": 0.5851 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5779 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5789 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6011 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.85 + "score": 0.6956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7434 + "score": 0.6747 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7508 + "score": 0.5988 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5851 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5779 + "score": 0.7847 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6011 + "score": 0.5789 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6956 + "score": 0.85 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6747 + "score": 0.7434 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5988 + "score": 0.7508 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/huggingfacetb_smollm2-135m-instruct.json b/data/models/huggingfacetb_smollm2-135m-instruct.json index 7413ccc13689271f5c637f0314f401eea3acbcec..4930fd119b327af5dcaff7489655e3b5a735f796 100644 --- a/data/models/huggingfacetb_smollm2-135m-instruct.json +++ b/data/models/huggingfacetb_smollm2-135m-instruct.json @@ -5,7 +5,7 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "0.135" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0593 + "score": 0.2883 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3135 + "score": 0.3124 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0144 + "score": 0.003 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2341 + "score": 0.2357 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3871 + "score": 0.3662 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1092 + "score": 0.1115 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2883 + "score": 0.0593 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3124 + "score": 0.3135 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.003 + "score": 0.0144 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2357 + "score": 0.2341 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3662 + "score": 0.3871 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1115 + "score": 0.1092 } } ], diff --git a/data/models/icefog72_icesakev6rp-7b.json b/data/models/icefog72_icesakev6rp-7b.json new file mode 100644 index 0000000000000000000000000000000000000000..f3267e4e350d12bc0c15245dbc590141fe9832d9 --- /dev/null +++ b/data/models/icefog72_icesakev6rp-7b.json @@ -0,0 +1,145 @@ +{ + "model_info": { + "name": "IceSakeV6RP-7b", + "id": "icefog72/IceSakeV6RP-7b", + "developer": "icefog72", + "inference_platform": "unknown", + "additional_details": { + "precision": "float16", + "architecture": "MistralForCausalLM", + "params_billions": "7.242" + } + }, + "evaluations": [ + { + "evaluation_id": "hfopenllm_v2/icefog72_IceSakeV6RP-7b/1773936498.240187", + "retrieved_timestamp": "1773936498.240187", + "source_metadata": { + "source_name": "HF Open LLM v2", + "source_type": "documentation", + "source_organization_name": "Hugging Face", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "lm-evaluation-harness", + "version": "0.4.0", + "additional_details": { + "fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess" + } + }, + "benchmark": "hfopenllm_v2", + "evaluation_results": [ + { + "evaluation_name": "IFEval", + "source_data": { + "dataset_name": "IFEval", + "source_type": "hf_dataset", + "hf_repo": "google/IFEval" + }, + "metric_config": { + "evaluation_description": "Accuracy on IFEval", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5033 + } + }, + { + "evaluation_name": "BBH", + "source_data": { + "dataset_name": "BBH", + "source_type": "hf_dataset", + "hf_repo": "SaylorTwift/bbh" + }, + "metric_config": { + "evaluation_description": "Accuracy on BBH", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.4976 + } + }, + { + "evaluation_name": "MATH Level 5", + "source_data": { + "dataset_name": "MATH Level 5", + "source_type": "hf_dataset", + "hf_repo": "DigitalLearningGmbH/MATH-lighteval" + }, + "metric_config": { + "evaluation_description": "Exact Match on MATH Level 5", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.0619 + } + }, + { + "evaluation_name": "GPQA", + "source_data": { + "dataset_name": "GPQA", + "source_type": "hf_dataset", + "hf_repo": "Idavidrein/gpqa" + }, + "metric_config": { + "evaluation_description": "Accuracy on GPQA", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.2911 + } + }, + { + "evaluation_name": "MUSR", + "source_data": { + "dataset_name": "MUSR", + "source_type": "hf_dataset", + "hf_repo": "TAUR-Lab/MuSR" + }, + "metric_config": { + "evaluation_description": "Accuracy on MUSR", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.42 + } + }, + { + "evaluation_name": "MMLU-PRO", + "source_data": { + "dataset_name": "MMLU-PRO", + "source_type": "hf_dataset", + "hf_repo": "TIGER-Lab/MMLU-Pro" + }, + "metric_config": { + "evaluation_description": "Accuracy on MMLU-PRO", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3093 + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + } + ] +} \ No newline at end of file diff --git a/data/models/internlm_internlm2-1_8b-reward.json b/data/models/internlm_internlm2-1_8b-reward.json index fdd95af043dae0ccf6943006b6d82bc9a69847ba..db02a96edd7d4deec66c9db68ed23c5ecfdc96f9 100644 --- a/data/models/internlm_internlm2-1_8b-reward.json +++ b/data/models/internlm_internlm2-1_8b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", + "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3902 + "score": 0.8217 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2758 + "score": 0.9358 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.6623 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4426 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4711 + "score": 0.8724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.596 + "score": 0.3902 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1934 + "score": 0.2758 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8217 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9358 + "score": 0.4426 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6623 + "score": 0.4711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8724 + "score": 0.1934 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/internlm_internlm2-20b-reward.json b/data/models/internlm_internlm2-20b-reward.json index 4de166855af75438b275f87369424e25bd91920e..db57bc6ddd293d585b6bca7ac06b0b270dabb864 100644 --- a/data/models/internlm_internlm2-20b-reward.json +++ b/data/models/internlm_internlm2-20b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9016 + "score": 0.5628 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9888 + "score": 0.5558 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7654 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8946 + "score": 0.5738 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9576 + "score": 0.6111 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5628 + "score": 0.7253 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5558 + "score": 0.5483 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.9016 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5738 + "score": 0.9888 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6111 + "score": 0.7654 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7253 + "score": 0.8946 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5483 + "score": 0.9576 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/internlm_internlm2-7b-reward.json b/data/models/internlm_internlm2-7b-reward.json index 850f56827380f314c355f2c532b41bbbb69062e8..5907aad77cdf2d42a7b25e1a1a35520112be4497 100644 --- a/data/models/internlm_internlm2-7b-reward.json +++ b/data/models/internlm_internlm2-7b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/internlm_internlm2-7b-reward/1766412838.146816", + "evaluation_id": "reward-bench/internlm_internlm2-7b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5335 + "score": 0.8759 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4211 + "score": 0.9916 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4 + "score": 0.6952 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5628 + "score": 0.8716 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5956 + "score": 0.9453 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/internlm_internlm2-7b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7051 + "score": 0.5335 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5164 + "score": 0.4211 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/internlm_internlm2-7b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8759 + "score": 0.4 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9916 + "score": 0.5628 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6952 + "score": 0.5956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8716 + "score": 0.7051 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9453 + "score": 0.5164 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/isaak-carter_josiev4o-8b-stage1-v4.json b/data/models/isaak-carter_josiev4o-8b-stage1-v4.json index 3d8b7bc1575ec467cf84b27d25552c6269300054..4be5b49897aa6863276434fc48c4d6962992c15e 100644 --- a/data/models/isaak-carter_josiev4o-8b-stage1-v4.json +++ b/data/models/isaak-carter_josiev4o-8b-stage1-v4.json @@ -5,7 +5,7 @@ "developer": "Isaak-Carter", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2477 + "score": 0.2553 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4725 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0453 + "score": 0.0529 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2911 + "score": 0.2919 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3641 + "score": 0.3654 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3316 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2553 + "score": 0.2477 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4725 + "score": 0.4758 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0529 + "score": 0.0453 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2919 + "score": 0.2911 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3654 + "score": 0.3641 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3316 + "score": 0.3292 } } ], diff --git a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json index 7b050770aeb59148493da52d2a21dee32c7d2c89..4462dbdfa0461cc5c2fd5d7ac79308c4e0a0cd93 100644 --- a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json +++ b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json @@ -5,7 +5,7 @@ "developer": "LeroyDyer", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "7.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.3066 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4575 + "score": 0.4577 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3012 + "score": 0.2995 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4253 + "score": 0.4254 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2329 + "score": 0.2318 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3066 + "score": 0.3036 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4577 + "score": 0.4575 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2995 + "score": 0.3012 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4254 + "score": 0.4253 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2318 + "score": 0.2329 } } ], diff --git a/data/models/llmat_mistral-v0.3-7b-orpo.json b/data/models/llmat_mistral-v0.3-7b-orpo.json index 3a1b947d84c5d76bb2423237a20cbf150415592e..c2c9120f6bd010e2ff566effb35617377de554ee 100644 --- a/data/models/llmat_mistral-v0.3-7b-orpo.json +++ b/data/models/llmat_mistral-v0.3-7b-orpo.json @@ -5,7 +5,7 @@ "developer": "llmat", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "7.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.364 + "score": 0.377 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4005 + "score": 0.3978 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0015 + "score": 0.0242 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2693 + "score": 0.2668 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3529 + "score": 0.3555 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2301 + "score": 0.2278 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.377 + "score": 0.364 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3978 + "score": 0.4005 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0242 + "score": 0.0015 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2668 + "score": 0.2693 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3555 + "score": 0.3529 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2278 + "score": 0.2301 } } ], diff --git a/data/models/microsoft_phi-3-mini-4k-instruct.json b/data/models/microsoft_phi-3-mini-4k-instruct.json index f0214854e07bd3ad87162726609ad6d31e31396a..9787f21694f92686e734f02c35c722665943d7d4 100644 --- a/data/models/microsoft_phi-3-mini-4k-instruct.json +++ b/data/models/microsoft_phi-3-mini-4k-instruct.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Phi3ForCausalLM", "params_billions": "3.821" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5477 + "score": 0.5613 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5491 + "score": 0.5676 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1639 + "score": 0.1163 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3322 + "score": 0.3196 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4284 + "score": 0.395 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4022 + "score": 0.3866 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5613 + "score": 0.5477 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5676 + "score": 0.5491 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1163 + "score": 0.1639 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3196 + "score": 0.3322 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.395 + "score": 0.4284 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3866 + "score": 0.4022 } } ], diff --git a/data/models/microsoft_phi-4.json b/data/models/microsoft_phi-4.json index 32eb099ab1f2e61ac11a8ea3a80df6bc18e5c656..c9fca4565946dbb26825690fda8366b38bda6709 100644 --- a/data/models/microsoft_phi-4.json +++ b/data/models/microsoft_phi-4.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Phi3ForCausalLM", "params_billions": "14.66" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0585 + "score": 0.0488 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6691 + "score": 0.6703 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3165 + "score": 0.2787 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.406 + "score": 0.401 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5287 + "score": 0.5295 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0488 + "score": 0.0585 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6703 + "score": 0.6691 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2787 + "score": 0.3165 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.401 + "score": 0.406 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5295 + "score": 0.5287 } } ], diff --git a/data/models/mistralai_mixtral-8x7b-v0.1.json b/data/models/mistralai_mixtral-8x7b-v0.1.json index 9d997e3527157e47894ae0f49b424a1634297e78..c3ac844f0072de0a86748b754d6c90570298768c 100644 --- a/data/models/mistralai_mixtral-8x7b-v0.1.json +++ b/data/models/mistralai_mixtral-8x7b-v0.1.json @@ -5,7 +5,7 @@ "developer": "mistralai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MixtralForCausalLM", "params_billions": "46.703" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2415 + "score": 0.2326 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5087 + "score": 0.5098 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.102 + "score": 0.0937 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3205 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4321 + "score": 0.4413 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.385 + "score": 0.3871 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2326 + "score": 0.2415 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5087 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.102 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3205 + "score": 0.3138 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4413 + "score": 0.4321 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3871 + "score": 0.385 } } ], diff --git a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json index d443de39bb7ed82b00df80190432e583c21fd660..7ef165972eafdeec56d923c82e02fdbbc9479eac 100644 --- a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json +++ b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json @@ -5,7 +5,7 @@ "developer": "mlabonne", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7561 + "score": 0.4162 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5111 + "score": 0.5124 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0906 + "score": 0.0853 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3029 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.415 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3841 + "score": 0.3802 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.7561 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5124 + "score": 0.5111 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0853 + "score": 0.0906 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3029 + "score": 0.3062 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.415 + "score": 0.4019 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3841 } } ], diff --git a/data/models/multiple_multiple.json b/data/models/multiple_multiple.json index c7a1aa9eb1cb1dfc4d88c122729fc6bae4148158..01b06a48bec6c0ec600ef83ad482f6552d46d980 100644 --- a/data/models/multiple_multiple.json +++ b/data/models/multiple_multiple.json @@ -4,13 +4,13 @@ "id": "multiple/multiple", "developer": "Multiple", "additional_details": { - "agent_name": "Abacus AI Desktop", - "agent_organization": "Abacus.AI" + "agent_name": "Warp", + "agent_organization": "Warp" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4, + "score": 61.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2026-03-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.2, + "score": 71.0, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-07", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.0, + "score": 58.4, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/nazimali_mistral-nemo-kurdish-instruct.json b/data/models/nazimali_mistral-nemo-kurdish-instruct.json index 7bcf436e7a29205a95d8b228b8e52bcfb9264e7a..bf12d1ac4ce4431e1cc4657892f1de48dd5df10b 100644 --- a/data/models/nazimali_mistral-nemo-kurdish-instruct.json +++ b/data/models/nazimali_mistral-nemo-kurdish-instruct.json @@ -5,7 +5,7 @@ "developer": "nazimali", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "12.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4964 + "score": 0.486 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4699 + "score": 0.4721 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0045 + "score": 0.0846 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2844 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3979 + "score": 0.4006 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3063 + "score": 0.3087 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.486 + "score": 0.4964 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4721 + "score": 0.4699 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0846 + "score": 0.0045 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2844 + "score": 0.2827 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4006 + "score": 0.3979 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3063 } } ], diff --git a/data/models/nexusflow_starling-rm-34b.json b/data/models/nexusflow_starling-rm-34b.json index 8ab4f392fbf3047e7bcb88c6ed9f3a2b6d8e5a37..0373bd963cc9795a2ad38fc9da30f8417d64ff6d 100644 --- a/data/models/nexusflow_starling-rm-34b.json +++ b/data/models/nexusflow_starling-rm-34b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4553 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4589 + "score": 0.8133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3187 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6175 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7556 + "score": 0.877 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4808 + "score": 0.8845 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1004 + "score": 0.7137 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8133 + "score": 0.4553 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.4589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3187 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6175 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.877 + "score": 0.7556 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8845 + "score": 0.4808 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7137 + "score": 0.1004 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/nicolinho_qrm-llama3.1-8b-v2.json b/data/models/nicolinho_qrm-llama3.1-8b-v2.json index 0df8878cad15f33ea391f78f6a5406e147177591..71e586c5d191f366e8b76150e58f0f9807a69f6b 100644 --- a/data/models/nicolinho_qrm-llama3.1-8b-v2.json +++ b/data/models/nicolinho_qrm-llama3.1-8b-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7074 + "score": 0.9314 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6653 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4062 + "score": 0.8684 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.612 + "score": 0.9257 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9467 + "score": 0.9677 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8909 + "score": 0.7074 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7234 + "score": 0.6653 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9314 + "score": 0.4062 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.612 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8684 + "score": 0.9467 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9257 + "score": 0.8909 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9677 + "score": 0.7234 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json index f86ab61b521749eb0f8fa030a3c29f5a32d327a1..486ba5e61208261c68f73d7d2bf88d78b1b36131 100644 --- a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json +++ b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json @@ -5,7 +5,7 @@ "developer": "ontocord", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.759" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1162 + "score": 0.1128 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3184 + "score": 0.3171 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0113 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2634 + "score": 0.2685 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3447 + "score": 0.346 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1124 + "score": 0.1129 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1162 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3171 + "score": 0.3184 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0113 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2685 + "score": 0.2634 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.3447 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1124 } } ], diff --git a/data/models/openai_gpt-4o-2024-08-06.json b/data/models/openai_gpt-4o-2024-08-06.json index 4523783fb76c33e56985cf8706dfcde93c76d009..ca15abfce433f9cf3b132cc5fefadfcef719867a 100644 --- a/data/models/openai_gpt-4o-2024-08-06.json +++ b/data/models/openai_gpt-4o-2024-08-06.json @@ -1900,10 +1900,10 @@ } }, { - "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", + "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1922,128 +1922,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8673 + "score": 0.6493 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9609 + "score": 0.5684 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8811 + "score": 0.623 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8661 + "score": 0.8619 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6493 + "score": 0.7293 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2052,111 +2028,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5684 + "score": 0.7819 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.8673 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.623 + "score": 0.9609 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8619 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.8811 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7819 + "score": 0.8661 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openai_gpt-4o-mini-2024-07-18.json b/data/models/openai_gpt-4o-mini-2024-07-18.json index f38f4e1ddedf08dfff3e3ceeb97aeebb3dcce913..1b3fb4c30102ee1f603e3640bb4c4b14c38b8cac 100644 --- a/data/models/openai_gpt-4o-mini-2024-07-18.json +++ b/data/models/openai_gpt-4o-mini-2024-07-18.json @@ -2124,10 +2124,10 @@ } }, { - "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2146,128 +2146,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8007 + "score": 0.5796 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.4105 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6075 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8081 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8374 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5796 + "score": 0.7414 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2276,111 +2252,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4105 + "score": 0.6962 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.8007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.6075 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7414 + "score": 0.8081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6962 + "score": 0.8374 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openai_gpt-5-2025-08-07.json b/data/models/openai_gpt-5-2025-08-07.json index 0853492fcc4bbd45f07185718454686350fe1be2..fdb97ce4111978e08938be252c1bd72645c11969 100644 --- a/data/models/openai_gpt-5-2025-08-07.json +++ b/data/models/openai_gpt-5-2025-08-07.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/openai_gpt-5-codex.json b/data/models/openai_gpt-5-codex.json index a875eb11356a27d3b1b0f42875c13e7560d24fec..bfb9d809d90faf4023c720a69d517cc43ddb6092 100644 --- a/data/models/openai_gpt-5-codex.json +++ b/data/models/openai_gpt-5-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Codex CLI", - "agent_organization": "OpenAI" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 44.3, + "score": 43.4, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 41.3, + "score": 44.3, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.4, + "score": 41.3, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-mini.json b/data/models/openai_gpt-5-mini.json index c2c25e33d8592b0cf47f06cc9166a5f66ec8f591..c2da3f31323642732ef7d6d76ecea92f3ae2ea0b 100644 --- a/data/models/openai_gpt-5-mini.json +++ b/data/models/openai_gpt-5-mini.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-mini", "developer": "OpenAI", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Codex CLI", + "agent_organization": "OpenAI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 22.2, + "score": 31.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 24.0, + "score": 29.2, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 24.0, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 31.9, + "score": 22.2, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-nano.json b/data/models/openai_gpt-5-nano.json index f1c6d8f032d8015d1380b26ffc2082a7ad57137f..17031b27e97ab53bd889b020db9c4f3335b70c14 100644 --- a/data/models/openai_gpt-5-nano.json +++ b/data/models/openai_gpt-5-nano.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-nano", "developer": "OpenAI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 9.9, + "score": 7.0, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 1.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 7.9, + "score": 9.9, "uncertainty": { "standard_error": { - "value": 1.9 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.0, + "score": 7.9, "uncertainty": { "standard_error": { "value": 1.9 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.1-codex.json b/data/models/openai_gpt-5.1-codex.json index 5ac777453649e155d2cf4ca5c0edb4c1fb8d16a4..d4b36a19fcc7086223ae8f218d54f2b1a6ca0184 100644 --- a/data/models/openai_gpt-5.1-codex.json +++ b/data/models/openai_gpt-5.1-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.1-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Crux", + "agent_organization": "Roam" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-17", + "evaluation_timestamp": "2025-11-16", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 57.8, "uncertainty": { "standard_error": { - "value": 3.2 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-16", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, + "score": 53.5, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 53.5, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.2-2025-12-11.json b/data/models/openai_gpt-5.2-2025-12-11.json index c3a672d6300dfa51d07ee396880a48a053600e4c..296a87eb83719a9fcbce40e624e36b0e1264dcb7 100644 --- a/data/models/openai_gpt-5.2-2025-12-11.json +++ b/data/models/openai_gpt-5.2-2025-12-11.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2-2025-12-11", "developer": "OpenAI", "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.071, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.55", + "total_run_cost": "55.03", + "average_steps": "51.59", + "percent_finished": "0.61" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.071, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "55.03", - "average_steps": "51.59", - "percent_finished": "0.61" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -138,8 +138,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -214,7 +214,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -261,8 +261,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -329,8 +329,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -342,8 +342,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -418,7 +418,7 @@ } }, { - "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.26, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.38", - "total_run_cost": "38.21", - "average_steps": "14.27", - "percent_finished": "1.0" + "average_agent_cost": "0.17", + "total_run_cost": "17.31", + "average_steps": "6.57", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -478,8 +478,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -622,7 +622,7 @@ } }, { - "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -654,23 +654,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.26, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.17", - "total_run_cost": "17.31", - "average_steps": "6.57", - "percent_finished": "0.99" + "average_agent_cost": "0.38", + "total_run_cost": "38.21", + "average_steps": "14.27", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -769,7 +769,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -801,14 +801,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5253, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.45", + "total_run_cost": "44.58", + "average_steps": "19.98", "percent_finished": "1.0" } }, @@ -816,8 +816,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -829,15 +829,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -869,14 +869,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5455, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.26", - "total_run_cost": "25.64", - "average_steps": "20.44", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -884,8 +884,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -897,15 +897,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -937,14 +937,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5253, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.45", - "total_run_cost": "44.58", - "average_steps": "19.98", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -952,8 +952,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -965,8 +965,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1041,7 +1041,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1073,14 +1073,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "5.77", - "average_steps": "11.4", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1088,8 +1088,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1101,15 +1101,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1122,33 +1122,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.11", + "total_run_cost": "5.77", + "average_steps": "11.4", "percent_finished": "1.0" } }, @@ -1156,8 +1156,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1169,15 +1169,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1209,14 +1209,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.6, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.29", + "total_run_cost": "15.28", + "average_steps": "10.68", "percent_finished": "1.0" } }, @@ -1224,8 +1224,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1237,15 +1237,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1277,14 +1277,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.23", - "average_steps": "10.18", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1292,8 +1292,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1305,15 +1305,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1326,33 +1326,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.5455, "uncertainty": { - "num_samples": 50 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.26", + "total_run_cost": "25.64", + "average_steps": "20.44", "percent_finished": "1.0" } }, @@ -1360,8 +1360,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1373,15 +1373,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1413,14 +1413,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6, + "score": 0.48, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.29", - "total_run_cost": "15.28", - "average_steps": "10.68", + "average_agent_cost": "0.21", + "total_run_cost": "11.23", + "average_steps": "10.18", "percent_finished": "1.0" } }, @@ -1428,8 +1428,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1441,15 +1441,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1481,23 +1481,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { "average_agent_cost": "0.11", - "total_run_cost": "11.54", - "average_steps": "9.55", - "percent_finished": "0.99" + "total_run_cost": "12.27", + "average_steps": "10.33", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1509,15 +1509,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1564,8 +1564,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1577,15 +1577,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1617,14 +1617,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", + "average_agent_cost": "0.25", + "total_run_cost": "26.27", + "average_steps": "11.08", "percent_finished": "1.0" } }, @@ -1632,8 +1632,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1645,15 +1645,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1685,23 +1685,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.12", - "total_run_cost": "12.63", - "average_steps": "9.92", - "percent_finished": "0.98" + "average_agent_cost": "0.11", + "total_run_cost": "11.54", + "average_steps": "9.55", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1713,15 +1713,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1734,33 +1734,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_telecom", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/telecom", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/telecom", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.71, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "26.27", - "average_steps": "11.08", + "average_agent_cost": "0.3", + "total_run_cost": "35.31", + "average_steps": "10.11", "percent_finished": "1.0" } }, @@ -1789,7 +1789,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1821,23 +1821,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.1", + "total_run_cost": "15.15", + "average_steps": "9.36", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1849,15 +1849,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1889,14 +1889,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.53, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "35.31", - "average_steps": "10.11", + "average_agent_cost": "0.15", + "total_run_cost": "18.88", + "average_steps": "9.92", "percent_finished": "1.0" } }, @@ -1904,8 +1904,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1917,15 +1917,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1957,23 +1957,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.53, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.15", - "total_run_cost": "18.88", - "average_steps": "9.92", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1985,15 +1985,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2025,23 +2025,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.1", - "total_run_cost": "15.15", - "average_steps": "9.36", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2053,15 +2053,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2074,42 +2074,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_telecom", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/telecom", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/telecom", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.12", + "total_run_cost": "12.63", + "average_steps": "9.92", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2121,8 +2121,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } diff --git a/data/models/openai_gpt-5.2.json b/data/models/openai_gpt-5.2.json index 1f373630f767c55e79d63767d1456fe1b49fa7e4..ff63f2052ea0b7c9a5cea40f5696ce46050721df 100644 --- a/data/models/openai_gpt-5.2.json +++ b/data/models/openai_gpt-5.2.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2", "developer": "OpenAI", "additional_details": { - "agent_name": "Droid", - "agent_organization": "Factory" + "agent_name": "Codex CLI", + "agent_organization": "OpenAI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.9, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,11 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, - "uncertainty": { - "standard_error": { - "value": 3.0 - }, - "num_samples": 435 - } + "score": 60.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +185,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.0, + "score": 64.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +212,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +226,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +250,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,11 +259,17 @@ "max_score": 100.0 }, "score_details": { - "score": 60.7 + "score": 54.0, + "uncertainty": { + "standard_error": { + "value": 2.9 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.3-codex.json b/data/models/openai_gpt-5.3-codex.json index 71af5af4fde2b8fad2d4e8ea66842bf1cb778d6d..65f8130b44c7c4318430a774c580739b4e8b1cab 100644 --- a/data/models/openai_gpt-5.3-codex.json +++ b/data/models/openai_gpt-5.3-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.3-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mux", + "agent_organization": "Coder" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.7, + "score": 74.6, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codebrain-1__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-10", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 75.1, + "score": 70.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-24", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 77.3, + "score": 64.7, "uncertainty": { "standard_error": { - "value": 2.2 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codebrain-1__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-10", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 70.3, + "score": 75.1, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2026-02-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.6, + "score": 77.3, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.json b/data/models/openai_gpt-5.json index 9d09df6c74b83cb2a4dd70946717742ac0b7c9f5..9005bf4e58d9a5ca4b1df67ba70d59695c801ffb 100644 --- a/data/models/openai_gpt-5.json +++ b/data/models/openai_gpt-5.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5", "developer": "OpenAI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.8, + "score": 33.9, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.2, + "score": 43.8, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.9, + "score": 35.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt_5.2.json b/data/models/openai_gpt_5.2.json index fec8e3d9659540e746cd368872cad0ec3c496416..e5de01feca478b0842db2b2533d006446492021b 100644 --- a/data/models/openai_gpt_5.2.json +++ b/data/models/openai_gpt_5.2.json @@ -7,10 +7,10 @@ }, "evaluations": [ { - "evaluation_id": "ace/openai_gpt-5.2/1773260200", + "evaluation_id": "apex-agents/openai_gpt-5.2/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { - "source_name": "Mercor ACE Leaderboard", + "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", @@ -20,24 +20,24 @@ "name": "archipelago", "version": "1.0.0" }, - "benchmark": "ace", + "benchmark": "apex-agents", "evaluation_results": [ { - "evaluation_name": "Overall Score", + "evaluation_name": "Overall Pass@1", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Overall ACE score across all consumer-task domains.", + "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.515, + "score": 0.23, "uncertainty": { "confidence_interval": { "lower": -0.032, @@ -53,21 +53,28 @@ } }, { - "evaluation_name": "Food Score", + "evaluation_name": "Overall Pass@8", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Food domain score.", + "evaluation_description": "Overall Pass@8 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.65 + "score": 0.4, + "uncertainty": { + "confidence_interval": { + "lower": -0.044, + "upper": 0.044, + "method": "bootstrap" + } + } }, "generation_config": { "additional_details": { @@ -76,75 +83,44 @@ } }, { - "evaluation_name": "Gaming Score", + "evaluation_name": "Overall Mean Score", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Gaming domain score.", + "evaluation_description": "Overall mean rubric score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.578 + "score": 0.387 }, "generation_config": { "additional_details": { "run_setting": "High" } } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - }, - { - "evaluation_id": "apex-agents/openai_gpt-5.2/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor APEX-Agents Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "apex-agents", - "evaluation_results": [ + }, { - "evaluation_name": "Overall Pass@1", + "evaluation_name": "Investment Banking Pass@1", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", + "evaluation_description": "Investment banking world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.23, - "uncertainty": { - "confidence_interval": { - "lower": -0.032, - "upper": 0.032, - "method": "bootstrap" - } - } + "score": 0.273 }, "generation_config": { "additional_details": { @@ -153,28 +129,21 @@ } }, { - "evaluation_name": "Overall Pass@8", + "evaluation_name": "Management Consulting Pass@1", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Overall Pass@8 (dataset card / paper snapshot).", + "evaluation_description": "Management consulting world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.4, - "uncertainty": { - "confidence_interval": { - "lower": -0.044, - "upper": 0.044, - "method": "bootstrap" - } - } + "score": 0.227 }, "generation_config": { "additional_details": { @@ -183,21 +152,21 @@ } }, { - "evaluation_name": "Overall Mean Score", + "evaluation_name": "Corporate Law Pass@1", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Overall mean rubric score.", + "evaluation_description": "Corporate law world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.387 + "score": 0.189 }, "generation_config": { "additional_details": { @@ -206,44 +175,75 @@ } }, { - "evaluation_name": "Investment Banking Pass@1", + "evaluation_name": "Corporate Lawyer Mean Score", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Investment banking world Pass@1.", + "evaluation_description": "Corporate lawyer world mean score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.273 + "score": 0.443 }, "generation_config": { "additional_details": { "run_setting": "High" } } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + }, + { + "evaluation_id": "ace/openai_gpt-5.2/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ { - "evaluation_name": "Management Consulting Pass@1", + "evaluation_name": "Overall Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Management consulting world Pass@1.", + "evaluation_description": "Overall ACE score across all consumer-task domains.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.227 + "score": 0.515, + "uncertainty": { + "confidence_interval": { + "lower": -0.032, + "upper": 0.032, + "method": "bootstrap" + } + } }, "generation_config": { "additional_details": { @@ -252,21 +252,21 @@ } }, { - "evaluation_name": "Corporate Law Pass@1", + "evaluation_name": "Food Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Corporate law world Pass@1.", + "evaluation_description": "Food domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.189 + "score": 0.65 }, "generation_config": { "additional_details": { @@ -275,21 +275,21 @@ } }, { - "evaluation_name": "Corporate Lawyer Mean Score", + "evaluation_name": "Gaming Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Corporate lawyer world mean score.", + "evaluation_description": "Gaming domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.443 + "score": 0.578 }, "generation_config": { "additional_details": { diff --git a/data/models/openai_o4-mini-2025-04-16.json b/data/models/openai_o4-mini-2025-04-16.json index 751e567f80107ae7403958c0915864809e731407..2445a24c2417bbf08441a2388a2fba31fa9d6f49 100644 --- a/data/models/openai_o4-mini-2025-04-16.json +++ b/data/models/openai_o4-mini-2025-04-16.json @@ -749,13 +749,13 @@ } }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", - "retrieved_timestamp": "1760492095.8105888", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", + "retrieved_timestamp": "1770683238.099205", "source_metadata": { - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation" + "source_type": "documentation", + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party" }, "eval_library": { "name": "unknown", @@ -765,62 +765,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", - "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.014084507042253521 - }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Medium Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Hard Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.30985915492957744 - }, + "score": 0.0143 + } + }, + { + "evaluation_name": "Medium Problems", "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Easy Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.8873239436619719 - }, + "score": 0.2923 + } + }, + { + "evaluation_name": "Easy Problems", "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] + }, + "metric_config": { + "evaluation_description": "Pass@1 on Easy Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.8571 } } ], @@ -828,13 +828,13 @@ "generation_config": null }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", - "retrieved_timestamp": "1770683238.099205", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", + "retrieved_timestamp": "1760492095.8105888", "source_metadata": { + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation", - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party" + "source_type": "documentation" }, "eval_library": { "name": "unknown", @@ -844,62 +844,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", + "metric_config": { + "evaluation_description": "Pass@1 on Hard Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.014084507042253521 + }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Medium Problems", "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.0143 - } - }, - { - "evaluation_name": "Medium Problems", + "score": 0.30985915492957744 + }, "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Easy Problems", "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Easy Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.2923 - } - }, - { - "evaluation_name": "Easy Problems", + "score": 0.8873239436619719 + }, "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] - }, - "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.8571 } } ], diff --git a/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json b/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json index cdfb6729d8ee2c00438e79bd148b5b9825fbce54..34afb9663c74112d98337da6fd83b3594c465f50 100644 --- a/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json +++ b/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2653 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3979 + "score": 0.615 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2875 + "score": 0.9246 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.377 + "score": 0.3728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.5446 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1535 + "score": 0.5855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.047 + "score": 0.6801 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.615 + "score": 0.2653 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9246 + "score": 0.3979 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.2875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.377 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5446 + "score": 0.3289 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5855 + "score": 0.1535 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6801 + "score": 0.047 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openassistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5.json b/data/models/openassistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5.json index 4f200b5bed952e0790f7951e7076e297510a1c46..d028cad4784e7392a6a614670c5c27a9b8900f31 100644 --- a/data/models/openassistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5.json +++ b/data/models/openassistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2648 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3179 + "score": 0.6901 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2625 + "score": 0.8855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3934 + "score": 0.4868 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3244 + "score": 0.6311 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2707 + "score": 0.7752 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0198 + "score": 0.6533 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2.1-pythia-1.4b-epoch-2.5/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6901 + "score": 0.2648 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8855 + "score": 0.3179 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4868 + "score": 0.2625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3934 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6311 + "score": 0.3244 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7752 + "score": 0.2707 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6533 + "score": 0.0198 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openassistant_reward-model-deberta-v3-large-v2.json b/data/models/openassistant_reward-model-deberta-v3-large-v2.json index cf1ba02f6dd573bf3e0770ff614660b73e92dbb8..b28ca5c5700af0f5cf22d77dcb1c4fea033ecc2d 100644 --- a/data/models/openassistant_reward-model-deberta-v3-large-v2.json +++ b/data/models/openassistant_reward-model-deberta-v3-large-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.32 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3853 + "score": 0.6126 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2687 + "score": 0.8939 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.4518 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3667 + "score": 0.7338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.3855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.12 + "score": 0.5836 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6126 + "score": 0.32 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8939 + "score": 0.3853 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4518 + "score": 0.2687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5027 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7338 + "score": 0.3667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3855 + "score": 0.2768 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5836 + "score": 0.12 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openbmb_eurus-rm-7b.json b/data/models/openbmb_eurus-rm-7b.json index e1154a660c89f219cec7ec844602d9d8fdfa08ad..44637ca17276f082725f61866dcfa3229cc54274 100644 --- a/data/models/openbmb_eurus-rm-7b.json +++ b/data/models/openbmb_eurus-rm-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5806 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6 + "score": 0.8159 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9804 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5683 + "score": 0.6557 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6267 + "score": 0.8135 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7475 + "score": 0.8633 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5972 + "score": 0.7172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8159 + "score": 0.5806 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9804 + "score": 0.6 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5683 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8135 + "score": 0.6267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8633 + "score": 0.7475 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7172 + "score": 0.5972 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openbmb_ultrarm-13b.json b/data/models/openbmb_ultrarm-13b.json index c52a509adb327ccd9798d5a844c78601940ebb17..84bdd483e26f91976f0250093c6ea14b4f0ff97c 100644 --- a/data/models/openbmb_ultrarm-13b.json +++ b/data/models/openbmb_ultrarm-13b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4683 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5063 + "score": 0.6903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5519 + "score": 0.5548 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5089 + "score": 0.5986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6081 + "score": 0.6244 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.7294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6903 + "score": 0.4683 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.5063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5548 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5519 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5986 + "score": 0.5089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6244 + "score": 0.6081 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7294 + "score": 0.3036 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v1.0-reward.json b/data/models/pku-alignment_beaver-7b-v1.0-reward.json index adf890d6a20deeb5311651870c4e5638866b9733..ee66fd95461643d78856676dc7b45eb953fc109f 100644 --- a/data/models/pku-alignment_beaver-7b-v1.0-reward.json +++ b/data/models/pku-alignment_beaver-7b-v1.0-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.1606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2105 + "score": 0.4727 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2938 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2623 + "score": 0.2873 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1422 + "score": 0.3757 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0646 + "score": 0.346 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.5993 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4727 + "score": 0.1606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.2105 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2873 + "score": 0.2938 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.2623 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3757 + "score": 0.1422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.0646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5993 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/primeintellect_intellect-1.json b/data/models/primeintellect_intellect-1.json index 7d9ec915394b34f60309abc9ed073a5aa8bce5ab..b8fa0710a43fde559062289de677b131f6527599 100644 --- a/data/models/primeintellect_intellect-1.json +++ b/data/models/primeintellect_intellect-1.json @@ -5,7 +5,7 @@ "developer": "PrimeIntellect", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "10.211" } @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.274 + "score": 0.276 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.25 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3753 + "score": 0.3339 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.112 + "score": 0.1123 } } ], @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.274 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.25 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3339 + "score": 0.3753 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1123 + "score": 0.112 } } ], diff --git a/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json b/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json index e065032f9e09241e38d49be10f5ffa3d463b4775..a2239eea28a466521ab226132351f6bc4c958619 100644 --- a/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json +++ b/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json @@ -5,12 +5,142 @@ "developer": "princeton-nlp", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } }, "evaluations": [ + { + "evaluation_id": "hfopenllm_v2/princeton-nlp_Llama-3-8B-ProLong-512k-Instruct/1773936498.240187", + "retrieved_timestamp": "1773936498.240187", + "source_metadata": { + "source_name": "HF Open LLM v2", + "source_type": "documentation", + "source_organization_name": "Hugging Face", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "lm-evaluation-harness", + "version": "0.4.0", + "additional_details": { + "fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess" + } + }, + "benchmark": "hfopenllm_v2", + "evaluation_results": [ + { + "evaluation_name": "IFEval", + "source_data": { + "dataset_name": "IFEval", + "source_type": "hf_dataset", + "hf_repo": "google/IFEval" + }, + "metric_config": { + "evaluation_description": "Accuracy on IFEval", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3978 + } + }, + { + "evaluation_name": "BBH", + "source_data": { + "dataset_name": "BBH", + "source_type": "hf_dataset", + "hf_repo": "SaylorTwift/bbh" + }, + "metric_config": { + "evaluation_description": "Accuracy on BBH", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.4983 + } + }, + { + "evaluation_name": "MATH Level 5", + "source_data": { + "dataset_name": "MATH Level 5", + "source_type": "hf_dataset", + "hf_repo": "DigitalLearningGmbH/MATH-lighteval" + }, + "metric_config": { + "evaluation_description": "Exact Match on MATH Level 5", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.0582 + } + }, + { + "evaluation_name": "GPQA", + "source_data": { + "dataset_name": "GPQA", + "source_type": "hf_dataset", + "hf_repo": "Idavidrein/gpqa" + }, + "metric_config": { + "evaluation_description": "Accuracy on GPQA", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.281 + } + }, + { + "evaluation_name": "MUSR", + "source_data": { + "dataset_name": "MUSR", + "source_type": "hf_dataset", + "hf_repo": "TAUR-Lab/MuSR" + }, + "metric_config": { + "evaluation_description": "Accuracy on MUSR", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.425 + } + }, + { + "evaluation_name": "MMLU-PRO", + "source_data": { + "dataset_name": "MMLU-PRO", + "source_type": "hf_dataset", + "hf_repo": "TIGER-Lab/MMLU-Pro" + }, + "metric_config": { + "evaluation_description": "Accuracy on MMLU-PRO", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3246 + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, { "evaluation_id": "hfopenllm_v2/princeton-nlp_Llama-3-8B-ProLong-512k-Instruct/1773936498.240187", "retrieved_timestamp": "1773936498.240187", diff --git a/data/models/prithivmlmods_calcium-opus-14b-elite.json b/data/models/prithivmlmods_calcium-opus-14b-elite.json index 89a0dd9277acb0b40655f39809af8ab4d96076ef..746fd41957e7795e0f0f3751181ce9bfef3d8e70 100644 --- a/data/models/prithivmlmods_calcium-opus-14b-elite.json +++ b/data/models/prithivmlmods_calcium-opus-14b-elite.json @@ -5,7 +5,7 @@ "developer": "prithivMLmods", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.766" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6052 + "score": 0.6064 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6317 + "score": 0.6296 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.3708 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3742 + "score": 0.3733 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.486 + "score": 0.4873 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5302 + "score": 0.5307 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6064 + "score": 0.6052 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6296 + "score": 0.6317 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3708 + "score": 0.4789 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3742 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4873 + "score": 0.486 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5307 + "score": 0.5302 } } ], diff --git a/data/models/qingy2019_oracle-14b.json b/data/models/qingy2019_oracle-14b.json index 22bb2fc596a8b7a91dfafd1db7c00009286e84b4..8fc1dd082923a347043a18d5c18fb5d4904fae42 100644 --- a/data/models/qingy2019_oracle-14b.json +++ b/data/models/qingy2019_oracle-14b.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MixtralForCausalLM", "params_billions": "13.668" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2358 + "score": 0.2401 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4612 + "score": 0.4622 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0642 + "score": 0.0725 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2609 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3717 + "score": 0.3703 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2382 + "score": 0.2379 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2401 + "score": 0.2358 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4622 + "score": 0.4612 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0725 + "score": 0.0642 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2576 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3703 + "score": 0.3717 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2379 + "score": 0.2382 } } ], diff --git a/data/models/qingy2019_qwen2.5-math-14b-instruct.json b/data/models/qingy2019_qwen2.5-math-14b-instruct.json index 21a12461d0cce8974037575db54605a0da19b661..d07b0024c79975f453423bba6d277d902ff3056c 100644 --- a/data/models/qingy2019_qwen2.5-math-14b-instruct.json +++ b/data/models/qingy2019_qwen2.5-math-14b-instruct.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6005 + "score": 0.6066 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6356 + "score": 0.635 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2764 + "score": 0.3716 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3691 + "score": 0.3725 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5339 + "score": 0.5331 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.6005 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.635 + "score": 0.6356 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.2764 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3725 + "score": 0.3691 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5331 + "score": 0.5339 } } ], diff --git a/data/models/quazim0t0_odb-14b-sce.json b/data/models/quazim0t0_odb-14b-sce.json index 0894dd13c3bc882b56c1b90b77e354b0b8fcf1b3..b2854fae2e7ea9ad5a32eb4ec453ef9c1365f1b7 100644 --- a/data/models/quazim0t0_odb-14b-sce.json +++ b/data/models/quazim0t0_odb-14b-sce.json @@ -6,8 +6,8 @@ "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", - "architecture": "Unknown", - "params_billions": "0.0", + "architecture": "LlamaForCausalLM", + "params_billions": "14.66", "model_id_aliases": [ "Quazim0t0/ODB-14b-sce" ] @@ -15,7 +15,7 @@ }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14B-sce/1773936498.240187", + "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14b-sce/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -47,7 +47,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2922 + "score": 0.7016 } }, { @@ -65,7 +65,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6559 + "score": 0.6942 } }, { @@ -83,7 +83,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2545 + "score": 0.4116 } }, { @@ -101,7 +101,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.3624 } }, { @@ -119,7 +119,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3929 + "score": 0.4571 } }, { @@ -137,7 +137,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5207 + "score": 0.5411 } } ], @@ -145,7 +145,7 @@ "generation_config": null }, { - "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14b-sce/1773936498.240187", + "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14B-sce/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -177,7 +177,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7016 + "score": 0.2922 } }, { @@ -195,7 +195,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6942 + "score": 0.6559 } }, { @@ -213,7 +213,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4116 + "score": 0.2545 } }, { @@ -231,7 +231,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3624 + "score": 0.2659 } }, { @@ -249,7 +249,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4571 + "score": 0.3929 } }, { @@ -267,7 +267,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5411 + "score": 0.5207 } } ], diff --git a/data/models/qwen_qwen2.5-coder-7b-instruct.json b/data/models/qwen_qwen2.5-coder-7b-instruct.json index 0cdcd4044d4f077bb1870a78d07a83c3bbc7d0d0..96b0a8af52fb442affa2ad7ba1ed017d80ec2231 100644 --- a/data/models/qwen_qwen2.5-coder-7b-instruct.json +++ b/data/models/qwen_qwen2.5-coder-7b-instruct.json @@ -5,7 +5,7 @@ "developer": "Qwen", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6147 + "score": 0.6101 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4999 + "score": 0.5008 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.3716 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2936 + "score": 0.2919 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.4073 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.3352 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6101 + "score": 0.6147 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5008 + "score": 0.4999 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.031 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2919 + "score": 0.2936 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4073 + "score": 0.4099 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3352 + "score": 0.3354 } } ], diff --git a/data/models/ray2333_grm-llama3-8b-distill.json b/data/models/ray2333_grm-llama3-8b-distill.json index a698e2dc0120e472c95dbd375c2ce72c243acf20..28bee941f1d39ce464ca73210ca8cbef67a5d082 100644 --- a/data/models/ray2333_grm-llama3-8b-distill.json +++ b/data/models/ray2333_grm-llama3-8b-distill.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8464 + "score": 0.589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5874 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6842 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5902 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9133 + "score": 0.6727 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7209 + "score": 0.5743 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.589 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5874 + "score": 0.8464 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5902 + "score": 0.6842 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7222 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6727 + "score": 0.9133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5743 + "score": 0.7209 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/ray2333_grm-llama3-8b-sftreg.json b/data/models/ray2333_grm-llama3-8b-sftreg.json index bd70639489991e3e8880cad041284119bd35ecc9..15f35c842dfd5169bfd85bae53f76a90d5850421 100644 --- a/data/models/ray2333_grm-llama3-8b-sftreg.json +++ b/data/models/ray2333_grm-llama3-8b-sftreg.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8542 + "score": 0.6089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.986 + "score": 0.6189 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5792 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8919 + "score": 0.7867 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9229 + "score": 0.6828 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7309 + "score": 0.5981 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6089 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6189 + "score": 0.8542 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5792 + "score": 0.6776 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7867 + "score": 0.8919 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6828 + "score": 0.9229 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5981 + "score": 0.7309 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json b/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json index 12ab734b7e3e0cffe5b67bb178df3f51d09ec598..9d341a3c927e4d25f7836c99807ee069a16cbedc 100644 --- a/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json +++ b/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json @@ -5,7 +5,7 @@ "developer": "rombodawg", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.397" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2566 + "score": 0.2595 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.39 + "score": 0.3884 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1208 + "score": 0.0914 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2626 + "score": 0.2743 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2741 + "score": 0.2719 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2595 + "score": 0.2566 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3884 + "score": 0.39 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0914 + "score": 0.1208 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.2626 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2719 + "score": 0.2741 } } ], diff --git a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json index 83ff30916d3f35c6333d5f652927c45d63836756..ecef2ed14b1e2210796bfe9e4943404f7ebd3b7c 100644 --- a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json +++ b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8338 + "score": 0.6292 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9944 + "score": 0.5916 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6513 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8644 + "score": 0.7051 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7492 + "score": 0.6647 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6292 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5916 + "score": 0.8338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9944 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.6513 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7051 + "score": 0.8644 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6647 + "score": 0.7492 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json index 4a85b178092e7e5acaf395eb054cb8adc61e2891..988f599afd76311b87e49f16160d0ca534436cdd 100644 --- a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json +++ b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9499 + "score": 0.7249 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.7558 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9079 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9378 + "score": 0.6448 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9903 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7249 + "score": 0.9131 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7558 + "score": 0.7633 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9499 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6448 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.9079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9131 + "score": 0.9378 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7633 + "score": 0.9903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json b/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json index 4109d07feac30d842a656539e17ae764cf441c79..52952d2b07ce3bbc9c007749585e45b758ae4213 100644 --- a/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json +++ b/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json @@ -142,10 +142,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -164,104 +164,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7531 + "score": 0.9426 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7674 + "score": 0.9609 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.8991 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6721 + "score": 0.9297 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9689 + "score": 0.9807 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9172 + "score": 0.7531 }, "source_data": { "dataset_name": "RewardBench 2", @@ -270,135 +294,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8182 + "score": 0.7674 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9426 + "score": 0.375 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9609 + "score": 0.6721 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8991 + "score": 0.9689 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9297 + "score": 0.9172 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9807 + "score": 0.8182 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/skywork_skywork-reward-llama-3.1-8b.json b/data/models/skywork_skywork-reward-llama-3.1-8b.json index d700e3f49377a01e2f9d11b4773879e002a65e98..dedd0015bc30c7a59cc8db5e1fbdb3b6b6cbc978 100644 --- a/data/models/skywork_skywork-reward-llama-3.1-8b.json +++ b/data/models/skywork_skywork-reward-llama-3.1-8b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7314 + "score": 0.9252 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6989 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.425 + "score": 0.8728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.9081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9333 + "score": 0.962 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9616 + "score": 0.7314 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.741 + "score": 0.6989 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9252 + "score": 0.425 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.6284 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8728 + "score": 0.9333 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9081 + "score": 0.9616 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.962 + "score": 0.741 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json index 5d50cd0f56f0ac33470efc8c81f51c02a702598e..7fb384620bce09b34b0d31ff763d2b1e971d0558 100644 --- a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json +++ b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json @@ -5,7 +5,7 @@ "developer": "tanliboy", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "9.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4501 + "score": 0.1829 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5472 + "score": 0.5488 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0944 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3104 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4017 + "score": 0.4056 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3792 + "score": 0.3805 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1829 + "score": 0.4501 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5488 + "score": 0.5472 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0944 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.3138 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4056 + "score": 0.4017 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3792 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-cobalt.json b/data/models/valiantlabs_llama3.1-8b-cobalt.json index e1d1a85c95f8deab607165ea40b9f93f0dba9c6d..c47cbaf67f9d28f7ee6c973b80a6e624aec4a0cc 100644 --- a/data/models/valiantlabs_llama3.1-8b-cobalt.json +++ b/data/models/valiantlabs_llama3.1-8b-cobalt.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7168 + "score": 0.3496 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4911 + "score": 0.4947 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1533 + "score": 0.1269 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3037 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3512 + "score": 0.3959 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3663 + "score": 0.3644 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3496 + "score": 0.7168 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4947 + "score": 0.4911 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1269 + "score": 0.1533 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3037 + "score": 0.2861 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3512 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3644 + "score": 0.3663 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json index f3e37b204fa779a9e21a0521a813b464c3fe641b..0736460b872bea97a209be68cef1f113fa7d9f3d 100644 --- a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json +++ b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6496 + "score": 0.2678 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4774 + "score": 0.4429 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0566 + "score": 0.0521 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.302 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3909 + "score": 0.3959 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3382 + "score": 0.2927 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2678 + "score": 0.6496 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4429 + "score": 0.4774 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0566 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.302 + "score": 0.3104 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3909 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2927 + "score": 0.3382 } } ], diff --git a/data/models/xai_grok-4-0709.json b/data/models/xai_grok-4-0709.json index ed63df2424f42d0db9790c340d534a65bcfb0c8a..c9c4c964a3fc0712105da9d9e183b32a52cc54bc 100644 --- a/data/models/xai_grok-4-0709.json +++ b/data/models/xai_grok-4-0709.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/xai_grok-4.json b/data/models/xai_grok-4.json index 8fb7b1b24e602a14c14ed4217ea5e7cd9fdf0b69..1a7ae0a5e37faeab9841b8816b2f9d9332f7d639 100644 --- a/data/models/xai_grok-4.json +++ b/data/models/xai_grok-4.json @@ -4,13 +4,13 @@ "id": "xai/grok-4", "developer": "xAI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.1, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.4, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 23.1, + "score": 25.4, "uncertainty": { "standard_error": { "value": 2.9 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/yam-peleg_hebrew-mistral-7b-200k.json b/data/models/yam-peleg_hebrew-mistral-7b-200k.json index baae674a05af479a549c721b22d1c0240e173193..1a70a6eb9b375af7d3ae622e407043222ccb78a0 100644 --- a/data/models/yam-peleg_hebrew-mistral-7b-200k.json +++ b/data/models/yam-peleg_hebrew-mistral-7b-200k.json @@ -5,7 +5,7 @@ "developer": "yam-peleg", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "7.504" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1856 + "score": 0.177 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.3411 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0234 + "score": 0.031 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3765 + "score": 0.374 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2573 + "score": 0.2529 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.177 + "score": 0.1856 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3411 + "score": 0.4149 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.0234 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.276 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.374 + "score": 0.3765 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2529 + "score": 0.2573 } } ], diff --git a/data/models/ycros_bagelmisterytour-v2-8x7b.json b/data/models/ycros_bagelmisterytour-v2-8x7b.json index ba69aabd10b1f09ccc48e0969d876027b03e3a4b..c7b7f840ab350df665db2f8b289c03a4556651c3 100644 --- a/data/models/ycros_bagelmisterytour-v2-8x7b.json +++ b/data/models/ycros_bagelmisterytour-v2-8x7b.json @@ -5,7 +5,7 @@ "developer": "ycros", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MixtralForCausalLM", "params_billions": "46.703" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6262 + "score": 0.5994 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5142 + "score": 0.5159 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.0785 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3079 + "score": 0.3045 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4138 + "score": 0.4203 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3481 + "score": 0.3473 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5994 + "score": 0.6262 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5159 + "score": 0.5142 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0785 + "score": 0.0937 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3045 + "score": 0.3079 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4203 + "score": 0.4138 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3473 + "score": 0.3481 } } ], diff --git a/data/models/zhipu-ai_glm-4.7.json b/data/models/zhipu-ai_glm-4.7.json index 9ffeed9fa5797e0518c941376cc795f418b5ed1c..ef6efd4fd9091259b58749fefdf86e2517723096 100644 --- a/data/models/zhipu-ai_glm-4.7.json +++ b/data/models/zhipu-ai_glm-4.7.json @@ -4,13 +4,13 @@ "id": "zhipu-ai/glm-4.7", "developer": "Z-AI", "additional_details": { - "agent_name": "Crux", - "agent_organization": "Roam" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/crux__glm-4.7/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__glm-4.7/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-08", + "evaluation_timestamp": "2026-01-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.3, + "score": 33.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__glm-4.7/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__glm-4.7/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-28", + "evaluation_timestamp": "2026-02-08", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.4, + "score": 33.3, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ {