diff --git a/data/benchmarks/appworld_test_normal.json b/data/benchmarks/appworld_test_normal.json index f3651490370442b9e2747ece9b07271eac5de3ad..4b1f2e226ce0f177163e0b4f9252a3d5461ce07f 100644 --- a/data/benchmarks/appworld_test_normal.json +++ b/data/benchmarks/appworld_test_normal.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "appworld/test_normal": 0.64 + "appworld/test_normal": 0.66 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "appworld/test_normal": 0.13 + "appworld/test_normal": 0.36 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "appworld/test_normal": 0.0 + "appworld/test_normal": 0.071 } } ] diff --git a/data/benchmarks/browsecompplus.json b/data/benchmarks/browsecompplus.json index 130d44f42fc301b55e5c6bce906e1c1263721368..613b29235acc848a4b6fbc21b6e89e5e195bd256 100644 --- a/data/benchmarks/browsecompplus.json +++ b/data/benchmarks/browsecompplus.json @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "browsecompplus": 0.48 + "browsecompplus": 0.46 } } ] diff --git a/data/benchmarks/hfopenllm_v2.json b/data/benchmarks/hfopenllm_v2.json index 589e88495313e5113fd7851135b88baeeba06feb..f63d6b4e97df8ea6a39802305cc7efdb2683daed 100644 --- a/data/benchmarks/hfopenllm_v2.json +++ b/data/benchmarks/hfopenllm_v2.json @@ -1019,12 +1019,12 @@ "name": "Qwen2.5-1.5B-continuous-learnt", "developer": "AtAndDev", "scores": { - "IFEval": 0.4511, - "BBH": 0.4275, - "MATH Level 5": 0.1473, - "GPQA": 0.2701, - "MUSR": 0.3623, - "MMLU-PRO": 0.2806 + "IFEval": 0.4605, + "BBH": 0.4258, + "MATH Level 5": 0.0748, + "GPQA": 0.2659, + "MUSR": 0.3636, + "MMLU-PRO": 0.2812 } }, { @@ -1747,12 +1747,12 @@ "name": "NeuralDaredevil-SuperNova-Lite-7B-DARETIES-abliterated", "developer": "BoltMonkey", "scores": { - "IFEval": 0.459, - "BBH": 0.5185, - "MATH Level 5": 0.0937, - "GPQA": 0.2743, - "MUSR": 0.4083, - "MMLU-PRO": 0.3631 + "IFEval": 0.7999, + "BBH": 0.5152, + "MATH Level 5": 0.1193, + "GPQA": 0.281, + "MUSR": 0.4019, + "MMLU-PRO": 0.3733 } }, { @@ -2176,12 +2176,12 @@ "name": "LION-Gemma-2b-dpo-v1.0", "developer": "Columbia-NLP", "scores": { - "IFEval": 0.3278, - "BBH": 0.392, - "MATH Level 5": 0.0431, - "GPQA": 0.2492, - "MUSR": 0.412, - "MMLU-PRO": 0.1666 + "IFEval": 0.3102, + "BBH": 0.3881, + "MATH Level 5": 0.0536, + "GPQA": 0.2534, + "MUSR": 0.4081, + "MMLU-PRO": 0.1665 } }, { @@ -3125,12 +3125,12 @@ "name": "DocumentCogito", "developer": "Daemontatox", "scores": { - "IFEval": 0.777, - "BBH": 0.5187, - "MATH Level 5": 0.2198, - "GPQA": 0.2936, - "MUSR": 0.3911, - "MMLU-PRO": 0.3738 + "IFEval": 0.5064, + "BBH": 0.5112, + "MATH Level 5": 0.1631, + "GPQA": 0.3163, + "MUSR": 0.3973, + "MMLU-PRO": 0.3802 } }, { @@ -3229,12 +3229,12 @@ "name": "PathfinderAI", "developer": "Daemontatox", "scores": { - "IFEval": 0.4855, - "BBH": 0.6627, - "MATH Level 5": 0.4841, - "GPQA": 0.3096, - "MUSR": 0.4256, - "MMLU-PRO": 0.5542 + "IFEval": 0.3745, + "BBH": 0.6668, + "MATH Level 5": 0.4758, + "GPQA": 0.3943, + "MUSR": 0.4858, + "MMLU-PRO": 0.5593 } }, { @@ -4321,12 +4321,12 @@ "name": "Llama-3.1-8b-ITA", "developer": "DeepMount00", "scores": { - "IFEval": 0.7917, - "BBH": 0.5109, - "MATH Level 5": 0.1088, - "GPQA": 0.2878, - "MUSR": 0.4136, - "MMLU-PRO": 0.3876 + "IFEval": 0.5365, + "BBH": 0.517, + "MATH Level 5": 0.1707, + "GPQA": 0.3062, + "MUSR": 0.4487, + "MMLU-PRO": 0.396 } }, { @@ -4646,12 +4646,12 @@ "name": "MN-12B-LilithFrame", "developer": "DoppelReflEx", "scores": { - "IFEval": 0.451, - "BBH": 0.4944, - "MATH Level 5": 0.1156, - "GPQA": 0.3196, - "MUSR": 0.3896, - "MMLU-PRO": 0.3256 + "IFEval": 0.436, + "BBH": 0.4956, + "MATH Level 5": 0.0589, + "GPQA": 0.3205, + "MUSR": 0.3843, + "MMLU-PRO": 0.3237 } }, { @@ -7025,12 +7025,12 @@ "name": "Fireball-Meta-Llama-3.1-8B-Instruct-Agent-0.003-128K-code-ds-auto", "developer": "EpistemeAI", "scores": { - "IFEval": 0.7207, - "BBH": 0.461, - "MATH Level 5": 0.1314, - "GPQA": 0.2701, - "MUSR": 0.3432, - "MMLU-PRO": 0.3354 + "IFEval": 0.7305, + "BBH": 0.4649, + "MATH Level 5": 0.1397, + "GPQA": 0.2659, + "MUSR": 0.3209, + "MMLU-PRO": 0.348 } }, { @@ -7675,12 +7675,12 @@ "name": "Herplete-LLM-Llama-3.1-8b", "developer": "Etherll", "scores": { - "IFEval": 0.4672, - "BBH": 0.5013, - "MATH Level 5": 0.0279, - "GPQA": 0.2861, - "MUSR": 0.386, - "MMLU-PRO": 0.3482 + "IFEval": 0.6106, + "BBH": 0.5347, + "MATH Level 5": 0.1548, + "GPQA": 0.3146, + "MUSR": 0.3991, + "MMLU-PRO": 0.3752 } }, { @@ -8572,12 +8572,12 @@ "name": "josie-7b-v6.0-step2000", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.7598, - "BBH": 0.5107, - "MATH Level 5": 0.4237, - "GPQA": 0.2768, - "MUSR": 0.4539, - "MMLU-PRO": 0.4012 + "IFEval": 0.7628, + "BBH": 0.5098, + "MATH Level 5": 0.0, + "GPQA": 0.2802, + "MUSR": 0.4579, + "MMLU-PRO": 0.4033 } }, { @@ -8728,12 +8728,12 @@ "name": "Gemma-Ko-Merge-PEFT", "developer": "Gunulhona", "scores": { - "IFEval": 0.4441, - "BBH": 0.4863, + "IFEval": 0.288, + "BBH": 0.5154, "MATH Level 5": 0.0, - "GPQA": 0.307, - "MUSR": 0.3986, - "MMLU-PRO": 0.3098 + "GPQA": 0.3247, + "MUSR": 0.408, + "MMLU-PRO": 0.3817 } }, { @@ -9170,12 +9170,12 @@ "name": "SmolLM2-360M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.3842, - "BBH": 0.3144, - "MATH Level 5": 0.0151, - "GPQA": 0.255, - "MUSR": 0.3461, - "MMLU-PRO": 0.1117 + "IFEval": 0.083, + "BBH": 0.3053, + "MATH Level 5": 0.0083, + "GPQA": 0.2651, + "MUSR": 0.3423, + "MMLU-PRO": 0.1126 } }, { @@ -9378,12 +9378,12 @@ "name": "JOSIEv4o-8b-stage1-v4", "developer": "Isaak-Carter", "scores": { - "IFEval": 0.2477, - "BBH": 0.4758, - "MATH Level 5": 0.0453, - "GPQA": 0.2911, - "MUSR": 0.3641, - "MMLU-PRO": 0.3292 + "IFEval": 0.2553, + "BBH": 0.4725, + "MATH Level 5": 0.0529, + "GPQA": 0.2919, + "MUSR": 0.3654, + "MMLU-PRO": 0.3316 } }, { @@ -14305,12 +14305,12 @@ "name": "Llama-3-8B-Magpie-Align-v0.1", "developer": "Magpie-Align", "scores": { - "IFEval": 0.4118, - "BBH": 0.4811, - "MATH Level 5": 0.034, - "GPQA": 0.2752, - "MUSR": 0.3047, - "MMLU-PRO": 0.3006 + "IFEval": 0.4027, + "BBH": 0.4789, + "MATH Level 5": 0.0461, + "GPQA": 0.2768, + "MUSR": 0.3087, + "MMLU-PRO": 0.3001 } }, { @@ -19986,12 +19986,12 @@ "name": "Replete-LLM-Qwen2-7b", "developer": "Replete-AI", "scores": { - "IFEval": 0.0932, - "BBH": 0.2977, + "IFEval": 0.0905, + "BBH": 0.2985, "MATH Level 5": 0.0, - "GPQA": 0.2475, - "MUSR": 0.3941, - "MMLU-PRO": 0.1157 + "GPQA": 0.2534, + "MUSR": 0.3848, + "MMLU-PRO": 0.1158 } }, { @@ -21130,12 +21130,12 @@ "name": "L3-70B-Euryale-v2.1", "developer": "Sao10K", "scores": { - "IFEval": 0.7281, - "BBH": 0.6503, - "MATH Level 5": 0.2243, + "IFEval": 0.7384, + "BBH": 0.6471, + "MATH Level 5": 0.2137, "GPQA": 0.3314, - "MUSR": 0.4196, - "MMLU-PRO": 0.5096 + "MUSR": 0.4209, + "MMLU-PRO": 0.5104 } }, { @@ -25121,12 +25121,12 @@ "name": "Llama3.1-8B-ShiningValiant2", "developer": "ValiantLabs", "scores": { - "IFEval": 0.6496, - "BBH": 0.4774, - "MATH Level 5": 0.0566, - "GPQA": 0.3104, - "MUSR": 0.3909, - "MMLU-PRO": 0.3382 + "IFEval": 0.2678, + "BBH": 0.4429, + "MATH Level 5": 0.0521, + "GPQA": 0.302, + "MUSR": 0.3959, + "MMLU-PRO": 0.2927 } }, { @@ -26603,12 +26603,12 @@ "name": "QAIMath-Qwen2.5-7B-TIES", "developer": "adriszmar", "scores": { - "IFEval": 0.1746, - "BBH": 0.3126, - "MATH Level 5": 0.0, - "GPQA": 0.245, - "MUSR": 0.4096, - "MMLU-PRO": 0.1087 + "IFEval": 0.1685, + "BBH": 0.3124, + "MATH Level 5": 0.0015, + "GPQA": 0.2492, + "MUSR": 0.3963, + "MMLU-PRO": 0.1066 } }, { @@ -26915,12 +26915,12 @@ "name": "Llama-3.1-Tulu-3-70B", "developer": "allenai", "scores": { - "IFEval": 0.8379, - "BBH": 0.6157, - "MATH Level 5": 0.3829, + "IFEval": 0.8291, + "BBH": 0.6164, + "MATH Level 5": 0.4502, "GPQA": 0.3733, - "MUSR": 0.4988, - "MMLU-PRO": 0.4656 + "MUSR": 0.4948, + "MMLU-PRO": 0.4645 } }, { @@ -26954,12 +26954,12 @@ "name": "Llama-3.1-Tulu-3-8B", "developer": "allenai", "scores": { - "IFEval": 0.8267, - "BBH": 0.405, - "MATH Level 5": 0.1964, - "GPQA": 0.2987, + "IFEval": 0.8255, + "BBH": 0.4061, + "MATH Level 5": 0.2115, + "GPQA": 0.297, "MUSR": 0.4175, - "MMLU-PRO": 0.2827 + "MMLU-PRO": 0.2821 } }, { @@ -28449,11 +28449,11 @@ "name": "AMD-Llama-135m", "developer": "amd", "scores": { - "IFEval": 0.1918, - "BBH": 0.2969, - "MATH Level 5": 0.0076, - "GPQA": 0.2584, - "MUSR": 0.3846, + "IFEval": 0.1842, + "BBH": 0.2974, + "MATH Level 5": 0.0053, + "GPQA": 0.2525, + "MUSR": 0.378, "MMLU-PRO": 0.1169 } }, @@ -28709,12 +28709,12 @@ "name": "Arcee-Spark", "developer": "arcee-ai", "scores": { - "IFEval": 0.5621, - "BBH": 0.5489, - "MATH Level 5": 0.2953, - "GPQA": 0.307, - "MUSR": 0.4021, - "MMLU-PRO": 0.3822 + "IFEval": 0.5718, + "BBH": 0.5481, + "MATH Level 5": 0.114, + "GPQA": 0.3062, + "MUSR": 0.4008, + "MMLU-PRO": 0.3813 } }, { @@ -31647,12 +31647,12 @@ "name": "dolphin-2.9.2-Phi-3-Medium-abliterated", "developer": "cognitivecomputations", "scores": { - "IFEval": 0.3613, - "BBH": 0.6123, - "MATH Level 5": 0.1239, - "GPQA": 0.328, - "MUSR": 0.4112, - "MMLU-PRO": 0.4494 + "IFEval": 0.4124, + "BBH": 0.6383, + "MATH Level 5": 0.182, + "GPQA": 0.3289, + "MUSR": 0.4349, + "MMLU-PRO": 0.4525 } }, { @@ -31790,12 +31790,12 @@ "name": "llama-43m-beta", "developer": "cpayne1303", "scores": { - "IFEval": 0.1916, - "BBH": 0.2977, - "MATH Level 5": 0.0, + "IFEval": 0.1949, + "BBH": 0.2965, + "MATH Level 5": 0.0045, "GPQA": 0.2685, - "MUSR": 0.3872, - "MMLU-PRO": 0.1132 + "MUSR": 0.3885, + "MMLU-PRO": 0.1111 } }, { @@ -33506,12 +33506,12 @@ "name": "TheBeagle-v2beta-32B-MGS", "developer": "fblgit", "scores": { - "IFEval": 0.5181, - "BBH": 0.7033, - "MATH Level 5": 0.4947, - "GPQA": 0.3826, - "MUSR": 0.5008, - "MMLU-PRO": 0.5915 + "IFEval": 0.4503, + "BBH": 0.7035, + "MATH Level 5": 0.3943, + "GPQA": 0.401, + "MUSR": 0.5021, + "MMLU-PRO": 0.5911 } }, { @@ -37705,12 +37705,12 @@ "name": "Kosmos-EVAA-Fusion-8B", "developer": "jaspionjader", "scores": { - "IFEval": 0.4345, - "BBH": 0.5419, - "MATH Level 5": 0.1292, - "GPQA": 0.3087, + "IFEval": 0.4418, + "BBH": 0.5406, + "MATH Level 5": 0.1352, + "GPQA": 0.3062, "MUSR": 0.4277, - "MMLU-PRO": 0.3854 + "MMLU-PRO": 0.386 } }, { @@ -41332,12 +41332,12 @@ "name": "BunderMaxx-0710", "developer": "kavonalds", "scores": { - "IFEval": 0.2701, - "BBH": 0.5566, + "IFEval": 0.3283, + "BBH": 0.6651, "MATH Level 5": 0.068, - "GPQA": 0.2802, - "MUSR": 0.3682, - "MMLU-PRO": 0.1449 + "GPQA": 0.2609, + "MUSR": 0.3393, + "MMLU-PRO": 0.1314 } }, { @@ -43893,12 +43893,12 @@ "name": "Meta-Llama-3-8B-Instruct", "developer": "meta-llama", "scores": { - "IFEval": 0.4782, - "BBH": 0.491, - "MATH Level 5": 0.0914, - "GPQA": 0.2928, - "MUSR": 0.3805, - "MMLU-PRO": 0.3591 + "IFEval": 0.7408, + "BBH": 0.4989, + "MATH Level 5": 0.0869, + "GPQA": 0.2592, + "MUSR": 0.3568, + "MMLU-PRO": 0.3664 } }, { @@ -44426,12 +44426,12 @@ "name": "Mistral-Small-Instruct-2409", "developer": "mistralai", "scores": { - "IFEval": 0.6283, - "BBH": 0.583, - "MATH Level 5": 0.2039, - "GPQA": 0.3331, - "MUSR": 0.4063, - "MMLU-PRO": 0.4099 + "IFEval": 0.667, + "BBH": 0.5213, + "MATH Level 5": 0.1435, + "GPQA": 0.3238, + "MUSR": 0.3632, + "MMLU-PRO": 0.396 } }, { @@ -44738,12 +44738,12 @@ "name": "NeuralDaredevil-8B-abliterated", "developer": "mlabonne", "scores": { - "IFEval": 0.7561, - "BBH": 0.5111, - "MATH Level 5": 0.0906, - "GPQA": 0.3062, - "MUSR": 0.4019, - "MMLU-PRO": 0.3841 + "IFEval": 0.4162, + "BBH": 0.5124, + "MATH Level 5": 0.0853, + "GPQA": 0.3029, + "MUSR": 0.415, + "MMLU-PRO": 0.3802 } }, { @@ -47611,12 +47611,12 @@ "name": "wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue", "developer": "ontocord", "scores": { - "IFEval": 0.1162, - "BBH": 0.3184, - "MATH Level 5": 0.0076, - "GPQA": 0.2634, - "MUSR": 0.3447, - "MMLU-PRO": 0.1124 + "IFEval": 0.1128, + "BBH": 0.3171, + "MATH Level 5": 0.0113, + "GPQA": 0.2685, + "MUSR": 0.346, + "MMLU-PRO": 0.1129 } }, { @@ -49444,12 +49444,12 @@ "name": "Calcium-Opus-14B-Elite", "developer": "prithivMLmods", "scores": { - "IFEval": 0.6052, - "BBH": 0.6317, - "MATH Level 5": 0.4789, - "GPQA": 0.3742, - "MUSR": 0.486, - "MMLU-PRO": 0.5302 + "IFEval": 0.6064, + "BBH": 0.6296, + "MATH Level 5": 0.3708, + "GPQA": 0.3733, + "MUSR": 0.4873, + "MMLU-PRO": 0.5307 } }, { @@ -50874,12 +50874,12 @@ "name": "Qwen2.5-Math-14B-Instruct", "developer": "qingy2019", "scores": { - "IFEval": 0.6005, - "BBH": 0.6356, - "MATH Level 5": 0.2764, - "GPQA": 0.3691, + "IFEval": 0.6066, + "BBH": 0.635, + "MATH Level 5": 0.3716, + "GPQA": 0.3725, "MUSR": 0.4757, - "MMLU-PRO": 0.5339 + "MMLU-PRO": 0.5331 } }, { @@ -51303,12 +51303,12 @@ "name": "Gemma-2-Ataraxy-Gemmasutra-9B-slerp", "developer": "recoilme", "scores": { - "IFEval": 0.2854, - "BBH": 0.5984, - "MATH Level 5": 0.1005, - "GPQA": 0.3297, - "MUSR": 0.4607, - "MMLU-PRO": 0.4162 + "IFEval": 0.7649, + "BBH": 0.5974, + "MATH Level 5": 0.0174, + "GPQA": 0.3305, + "MUSR": 0.4245, + "MMLU-PRO": 0.4207 } }, { @@ -51329,12 +51329,12 @@ "name": "recoilme-gemma-2-9B-v0.2", "developer": "recoilme", "scores": { - "IFEval": 0.2747, - "BBH": 0.6031, - "MATH Level 5": 0.0831, - "GPQA": 0.3305, - "MUSR": 0.4686, - "MMLU-PRO": 0.4122 + "IFEval": 0.7592, + "BBH": 0.6026, + "MATH Level 5": 0.0529, + "GPQA": 0.3289, + "MUSR": 0.4099, + "MMLU-PRO": 0.4163 } }, { @@ -51342,12 +51342,12 @@ "name": "recoilme-gemma-2-9B-v0.3", "developer": "recoilme", "scores": { - "IFEval": 0.7439, - "BBH": 0.5993, - "MATH Level 5": 0.0876, - "GPQA": 0.3238, - "MUSR": 0.4204, - "MMLU-PRO": 0.4072 + "IFEval": 0.5761, + "BBH": 0.602, + "MATH Level 5": 0.1888, + "GPQA": 0.3372, + "MUSR": 0.4632, + "MMLU-PRO": 0.4039 } }, { @@ -51459,12 +51459,12 @@ "name": "FineLlama-3.1-8B", "developer": "riaz", "scores": { - "IFEval": 0.4373, - "BBH": 0.4586, - "MATH Level 5": 0.0514, - "GPQA": 0.2752, - "MUSR": 0.3763, - "MMLU-PRO": 0.2964 + "IFEval": 0.4137, + "BBH": 0.4565, + "MATH Level 5": 0.0453, + "GPQA": 0.276, + "MUSR": 0.3776, + "MMLU-PRO": 0.2978 } }, { @@ -51602,12 +51602,12 @@ "name": "Rombos-LLM-V2.5.1-Qwen-3b", "developer": "rombodawg", "scores": { - "IFEval": 0.2566, - "BBH": 0.39, - "MATH Level 5": 0.1208, - "GPQA": 0.2626, + "IFEval": 0.2595, + "BBH": 0.3884, + "MATH Level 5": 0.0914, + "GPQA": 0.2743, "MUSR": 0.3991, - "MMLU-PRO": 0.2741 + "MMLU-PRO": 0.2719 } }, { @@ -54345,12 +54345,12 @@ "name": "lambda-gemma-2-9b-dpo", "developer": "tanliboy", "scores": { - "IFEval": 0.4501, - "BBH": 0.5472, - "MATH Level 5": 0.0944, - "GPQA": 0.3138, - "MUSR": 0.4017, - "MMLU-PRO": 0.3792 + "IFEval": 0.1829, + "BBH": 0.5488, + "MATH Level 5": 0.0, + "GPQA": 0.3104, + "MUSR": 0.4056, + "MMLU-PRO": 0.3805 } }, { @@ -56945,12 +56945,12 @@ "name": "Hebrew-Mistral-7B-200K", "developer": "yam-peleg", "scores": { - "IFEval": 0.1856, - "BBH": 0.4149, - "MATH Level 5": 0.0234, - "GPQA": 0.276, - "MUSR": 0.3765, - "MMLU-PRO": 0.2573 + "IFEval": 0.177, + "BBH": 0.3411, + "MATH Level 5": 0.031, + "GPQA": 0.2534, + "MUSR": 0.374, + "MMLU-PRO": 0.2529 } }, { diff --git a/data/benchmarks/reward-bench.json b/data/benchmarks/reward-bench.json index 435eb05406175ffe8530e3141464bcd0e3abcc31..083fee6e840e33c85c4fe87b781938b7ad720c00 100644 --- a/data/benchmarks/reward-bench.json +++ b/data/benchmarks/reward-bench.json @@ -453,16 +453,16 @@ "name": "LxzGordon/URM-LLaMa-3.1-8B", "developer": "LxzGordon", "scores": { - "Score": 0.9294, + "Score": 0.7394, + "Chat": 0.9553, + "Chat Hard": 0.8816, + "Safety": 0.9178, + "Reasoning": 0.9698, "Factuality": 0.6884, "Precise IF": 0.45, "Math": 0.6393, - "Safety": 0.9108, "Focus": 0.9758, - "Ties": 0.7653, - "Chat": 0.9553, - "Chat Hard": 0.8816, - "Reasoning": 0.9698 + "Ties": 0.7653 } }, { @@ -482,16 +482,16 @@ "name": "NCSOFT/Llama-3-OffsetBias-RM-8B", "developer": "NCSOFT", "scores": { - "Score": 0.8942, + "Score": 0.648, + "Chat": 0.9721, + "Chat Hard": 0.818, + "Safety": 0.7222, + "Reasoning": 0.9192, "Factuality": 0.6084, "Precise IF": 0.4, "Math": 0.5191, - "Safety": 0.8676, "Focus": 0.9596, - "Ties": 0.6786, - "Chat": 0.9721, - "Chat Hard": 0.818, - "Reasoning": 0.9192 + "Ties": 0.6786 } }, { @@ -499,17 +499,17 @@ "name": "Nexusflow/Starling-RM-34B", "developer": "Nexusflow", "scores": { - "Score": 0.4553, - "Chat": 0.9693, - "Chat Hard": 0.5724, - "Safety": 0.7556, - "Reasoning": 0.8845, - "Prior Sets (0.5 weight)": 0.7137, + "Score": 0.8133, "Factuality": 0.4589, "Precise IF": 0.3187, "Math": 0.6175, + "Safety": 0.877, "Focus": 0.4808, - "Ties": 0.1004 + "Ties": 0.1004, + "Chat": 0.9693, + "Chat Hard": 0.5724, + "Reasoning": 0.8845, + "Prior Sets (0.5 weight)": 0.7137 } }, { @@ -591,17 +591,17 @@ "name": "OpenAssistant/reward-model-deberta-v3-large-v2", "developer": "OpenAssistant", "scores": { - "Score": 0.32, - "Chat": 0.8939, - "Chat Hard": 0.4518, - "Safety": 0.3667, - "Reasoning": 0.3855, - "Prior Sets (0.5 weight)": 0.5836, + "Score": 0.6126, "Factuality": 0.3853, "Precise IF": 0.2687, "Math": 0.5027, + "Safety": 0.7338, "Focus": 0.2768, - "Ties": 0.12 + "Ties": 0.12, + "Chat": 0.8939, + "Chat Hard": 0.4518, + "Reasoning": 0.3855, + "Prior Sets (0.5 weight)": 0.5836 } }, { @@ -609,17 +609,17 @@ "name": "PKU-Alignment/beaver-7b-v1.0-cost", "developer": "PKU-Alignment", "scores": { - "Score": 0.3332, - "Chat": 0.6173, - "Chat Hard": 0.4232, - "Safety": 0.7589, - "Reasoning": 0.5482, - "Prior Sets (0.5 weight)": 0.57, + "Score": 0.5798, "Factuality": 0.3263, "Precise IF": 0.2313, "Math": 0.3989, + "Safety": 0.7351, "Focus": 0.2939, - "Ties": -0.01 + "Ties": -0.01, + "Chat": 0.6173, + "Chat Hard": 0.4232, + "Reasoning": 0.5482, + "Prior Sets (0.5 weight)": 0.57 } }, { @@ -904,16 +904,16 @@ "name": "Ray2333/GRM-Llama3-8B-rewardmodel-ft", "developer": "Ray2333", "scores": { - "Score": 0.9154, + "Score": 0.6766, + "Chat": 0.9553, + "Chat Hard": 0.8618, + "Safety": 0.9222, + "Reasoning": 0.9362, "Factuality": 0.6274, "Precise IF": 0.35, "Math": 0.5847, - "Safety": 0.9081, "Focus": 0.8929, - "Ties": 0.6824, - "Chat": 0.9553, - "Chat Hard": 0.8618, - "Reasoning": 0.9362 + "Ties": 0.6824 } }, { @@ -938,17 +938,17 @@ "name": "Ray2333/GRM-llama3-8B-distill", "developer": "Ray2333", "scores": { - "Score": 0.8464, + "Score": 0.589, + "Chat": 0.9832, + "Chat Hard": 0.6842, + "Safety": 0.7222, + "Reasoning": 0.9133, + "Prior Sets (0.5 weight)": 0.7209, "Factuality": 0.5874, "Precise IF": 0.3875, "Math": 0.5902, - "Safety": 0.8676, "Focus": 0.6727, - "Ties": 0.5743, - "Chat": 0.9832, - "Chat Hard": 0.6842, - "Reasoning": 0.9133, - "Prior Sets (0.5 weight)": 0.7209 + "Ties": 0.5743 } }, { @@ -956,17 +956,17 @@ "name": "Ray2333/GRM-llama3-8B-sftreg", "developer": "Ray2333", "scores": { - "Score": 0.8542, + "Score": 0.6089, + "Chat": 0.986, + "Chat Hard": 0.6776, + "Safety": 0.7867, + "Reasoning": 0.9229, + "Prior Sets (0.5 weight)": 0.7309, "Factuality": 0.6189, "Precise IF": 0.3875, "Math": 0.5792, - "Safety": 0.8919, "Focus": 0.6828, - "Ties": 0.5981, - "Chat": 0.986, - "Chat Hard": 0.6776, - "Reasoning": 0.9229, - "Prior Sets (0.5 weight)": 0.7309 + "Ties": 0.5981 } }, { @@ -1098,16 +1098,16 @@ "name": "ShikaiChen/LDL-Reward-Gemma-2-27B-v0.1", "developer": "ShikaiChen", "scores": { - "Score": 0.9499, + "Score": 0.7249, + "Chat": 0.9637, + "Chat Hard": 0.9079, + "Safety": 0.9222, + "Reasoning": 0.9903, "Factuality": 0.7558, "Precise IF": 0.35, "Math": 0.6448, - "Safety": 0.9378, "Focus": 0.9131, - "Ties": 0.7633, - "Chat": 0.9637, - "Chat Hard": 0.9079, - "Reasoning": 0.9903 + "Ties": 0.7633 } }, { @@ -1156,16 +1156,16 @@ "name": "Skywork/Skywork-Reward-Gemma-2-27B-v0.2", "developer": "Skywork", "scores": { - "Score": 0.7531, - "Chat": 0.9609, - "Chat Hard": 0.8991, - "Safety": 0.9689, - "Reasoning": 0.9807, + "Score": 0.9426, "Factuality": 0.7674, "Precise IF": 0.375, "Math": 0.6721, + "Safety": 0.9297, "Focus": 0.9172, - "Ties": 0.8182 + "Ties": 0.8182, + "Chat": 0.9609, + "Chat Hard": 0.8991, + "Reasoning": 0.9807 } }, { @@ -1379,9 +1379,9 @@ "name": "ai2/tulu-2-7b-rm-v0-nectar-binarized-3.8m-check...", "developer": "AI2", "scores": { - "Score": 0.7008, - "Chat": 0.9385, - "Chat Hard": 0.3882, + "Score": 0.6924, + "Chat": 0.9441, + "Chat Hard": 0.3575, "Safety": 0.7757 } }, @@ -1423,17 +1423,17 @@ "name": "allenai/Llama-3.1-70B-Instruct-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.9021, + "Score": 0.7606, + "Chat": 0.9665, + "Chat Hard": 0.8355, + "Safety": 0.8844, + "Reasoning": 0.8969, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.8126, "Precise IF": 0.4188, "Math": 0.6995, - "Safety": 0.9095, "Focus": 0.8646, - "Ties": 0.8835, - "Chat": 0.9665, - "Chat Hard": 0.8355, - "Reasoning": 0.8969, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.8835 } }, { @@ -1513,17 +1513,17 @@ "name": "allenai/Llama-3.1-Tulu-3-8B-RL-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8369, + "Score": 0.6871, + "Chat": 0.9469, + "Chat Hard": 0.7588, + "Safety": 0.8644, + "Reasoning": 0.7715, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.7642, "Precise IF": 0.4, "Math": 0.6175, - "Safety": 0.8703, "Focus": 0.8485, - "Ties": 0.6281, - "Chat": 0.9469, - "Chat Hard": 0.7588, - "Reasoning": 0.7715, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.6281 } }, { @@ -3504,16 +3504,16 @@ "name": "Claude 3 Opus 20240229", "developer": "Anthropic", "scores": { - "Score": 0.5744, - "Chat": 0.9469, - "Chat Hard": 0.6031, - "Safety": 0.8378, - "Reasoning": 0.7868, + "Score": 0.8008, "Factuality": 0.5389, "Precise IF": 0.3312, "Math": 0.5137, + "Safety": 0.8662, "Focus": 0.6646, - "Ties": 0.5601 + "Ties": 0.5601, + "Chat": 0.9469, + "Chat Hard": 0.6031, + "Reasoning": 0.7868 } }, { @@ -3766,17 +3766,17 @@ "name": "hendrydong/Mistral-RM-for-RAFT-GSHF-v0", "developer": "hendrydong", "scores": { - "Score": 0.7847, + "Score": 0.5851, + "Chat": 0.9832, + "Chat Hard": 0.5789, + "Safety": 0.6956, + "Reasoning": 0.7434, + "Prior Sets (0.5 weight)": 0.7508, "Factuality": 0.5779, "Precise IF": 0.3625, "Math": 0.6011, - "Safety": 0.85, "Focus": 0.6747, - "Ties": 0.5988, - "Chat": 0.9832, - "Chat Hard": 0.5789, - "Reasoning": 0.7434, - "Prior Sets (0.5 weight)": 0.7508 + "Ties": 0.5988 } }, { @@ -3784,16 +3784,16 @@ "name": "infly/INF-ORM-Llama3.1-70B", "developer": "infly", "scores": { - "Score": 0.7648, - "Chat": 0.9665, - "Chat Hard": 0.9101, - "Safety": 0.9644, - "Reasoning": 0.9912, + "Score": 0.9511, "Factuality": 0.7411, "Precise IF": 0.4188, "Math": 0.6995, + "Safety": 0.9365, "Focus": 0.903, - "Ties": 0.8622 + "Ties": 0.8622, + "Chat": 0.9665, + "Chat Hard": 0.9101, + "Reasoning": 0.9912 } }, { @@ -3801,16 +3801,16 @@ "name": "internlm/internlm2-1_8b-reward", "developer": "internlm", "scores": { - "Score": 0.3902, - "Chat": 0.9358, - "Chat Hard": 0.6623, - "Safety": 0.4711, - "Reasoning": 0.8724, + "Score": 0.8217, "Factuality": 0.2758, "Precise IF": 0.3625, "Math": 0.4426, + "Safety": 0.8162, "Focus": 0.596, - "Ties": 0.1934 + "Ties": 0.1934, + "Chat": 0.9358, + "Chat Hard": 0.6623, + "Reasoning": 0.8724 } }, { @@ -3818,16 +3818,16 @@ "name": "internlm/internlm2-20b-reward", "developer": "internlm", "scores": { - "Score": 0.9016, + "Score": 0.5628, + "Chat": 0.9888, + "Chat Hard": 0.7654, + "Safety": 0.6111, + "Reasoning": 0.9576, "Factuality": 0.5558, "Precise IF": 0.3625, "Math": 0.5738, - "Safety": 0.8946, "Focus": 0.7253, - "Ties": 0.5483, - "Chat": 0.9888, - "Chat Hard": 0.7654, - "Reasoning": 0.9576 + "Ties": 0.5483 } }, { @@ -4202,16 +4202,16 @@ "name": "GPT-4o 2024-08-06", "developer": "OpenAI", "scores": { - "Score": 0.8673, + "Score": 0.6493, + "Chat": 0.9609, + "Chat Hard": 0.761, + "Safety": 0.8619, + "Reasoning": 0.8661, "Factuality": 0.5684, "Precise IF": 0.3312, "Math": 0.623, - "Safety": 0.8811, "Focus": 0.7293, - "Ties": 0.7819, - "Chat": 0.9609, - "Chat Hard": 0.761, - "Reasoning": 0.8661 + "Ties": 0.7819 } }, { @@ -4219,16 +4219,16 @@ "name": "GPT-4o mini 2024-07-18", "developer": "OpenAI", "scores": { - "Score": 0.8007, + "Score": 0.5796, + "Chat": 0.9497, + "Chat Hard": 0.6075, + "Safety": 0.7667, + "Reasoning": 0.8374, "Factuality": 0.4105, "Precise IF": 0.3438, "Math": 0.5191, - "Safety": 0.8081, "Focus": 0.7414, - "Ties": 0.6962, - "Chat": 0.9497, - "Chat Hard": 0.6075, - "Reasoning": 0.8374 + "Ties": 0.6962 } }, { @@ -4249,17 +4249,17 @@ "name": "openbmb/Eurus-RM-7b", "developer": "openbmb", "scores": { - "Score": 0.5806, - "Chat": 0.9804, - "Chat Hard": 0.6557, - "Safety": 0.6267, - "Reasoning": 0.8633, - "Prior Sets (0.5 weight)": 0.7172, + "Score": 0.8159, "Factuality": 0.6, "Precise IF": 0.3438, "Math": 0.5683, + "Safety": 0.8135, "Focus": 0.7475, - "Ties": 0.5972 + "Ties": 0.5972, + "Chat": 0.9804, + "Chat Hard": 0.6557, + "Reasoning": 0.8633, + "Prior Sets (0.5 weight)": 0.7172 } }, { @@ -4510,17 +4510,17 @@ "name": "weqweasdas/RM-Gemma-7B", "developer": "weqweasdas", "scores": { - "Score": 0.4826, - "Chat": 0.9693, - "Chat Hard": 0.4978, - "Safety": 0.4822, - "Reasoning": 0.7362, - "Prior Sets (0.5 weight)": 0.7069, + "Score": 0.6967, "Factuality": 0.4926, "Precise IF": 0.3937, "Math": 0.6066, + "Safety": 0.5784, "Focus": 0.497, - "Ties": 0.4232 + "Ties": 0.4232, + "Chat": 0.9693, + "Chat Hard": 0.4978, + "Reasoning": 0.7362, + "Prior Sets (0.5 weight)": 0.7069 } }, { @@ -4559,17 +4559,17 @@ "name": "weqweasdas/hh_rlhf_rm_open_llama_3b", "developer": "weqweasdas", "scores": { - "Score": 0.2498, - "Chat": 0.8184, - "Chat Hard": 0.3728, - "Safety": 0.24, - "Reasoning": 0.3281, - "Prior Sets (0.5 weight)": 0.6564, + "Score": 0.5027, "Factuality": 0.3642, "Precise IF": 0.275, "Math": 0.3497, + "Safety": 0.4149, "Focus": 0.2384, - "Ties": 0.0315 + "Ties": 0.0315, + "Chat": 0.8184, + "Chat Hard": 0.3728, + "Reasoning": 0.3281, + "Prior Sets (0.5 weight)": 0.6564 } } ] diff --git a/data/benchmarks/swe-bench.json b/data/benchmarks/swe-bench.json index 093176be195c26e24230b48d3e9b488139a6d6bf..3b6c1cd3a38d41c01ec61c5d181997bdb0b7b011 100644 --- a/data/benchmarks/swe-bench.json +++ b/data/benchmarks/swe-bench.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "swe-bench": 0.8072 + "swe-bench": 0.65 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "swe-bench": 0.5455 + "swe-bench": 0.57 } } ] diff --git a/data/benchmarks/tau-bench-2_airline.json b/data/benchmarks/tau-bench-2_airline.json index 203ba536a38c5fd6f5a3b4f5a7c625bb98c214a5..3829696a07bc037642c69851a891e0aeb0e5febf 100644 --- a/data/benchmarks/tau-bench-2_airline.json +++ b/data/benchmarks/tau-bench-2_airline.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/airline": 0.66 + "tau-bench-2/airline": 0.72 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/airline": 0.7 + "tau-bench-2/airline": 0.68 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/airline": 0.48 + "tau-bench-2/airline": 0.6 } } ] diff --git a/data/benchmarks/tau-bench-2_retail.json b/data/benchmarks/tau-bench-2_retail.json index bb3c509d520d0e5ab84b4c3e44db8c84041b45fa..43bf4e72d939794b39171defb2c6b721324ab767 100644 --- a/data/benchmarks/tau-bench-2_retail.json +++ b/data/benchmarks/tau-bench-2_retail.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/retail": 0.7805 + "tau-bench-2/retail": 0.82 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/retail": 0.51 + "tau-bench-2/retail": 0.73 } } ] diff --git a/data/benchmarks/tau-bench-2_telecom.json b/data/benchmarks/tau-bench-2_telecom.json index 0371242e1ef28be51ed37461c45171b5f6b938db..05002e54c35c11c6e2530af6d100378b4038ab6c 100644 --- a/data/benchmarks/tau-bench-2_telecom.json +++ b/data/benchmarks/tau-bench-2_telecom.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/telecom": 0.6852 + "tau-bench-2/telecom": 0.73 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/telecom": 0.5354 + "tau-bench-2/telecom": 0.55 } } ] diff --git a/data/benchmarks/terminal-bench-2.0.json b/data/benchmarks/terminal-bench-2.0.json index a8e801f280cc4af65aa221436665986581643fe9..3f1515b599ae57812585a3e55eaf803056a58325 100644 --- a/data/benchmarks/terminal-bench-2.0.json +++ b/data/benchmarks/terminal-bench-2.0.json @@ -13,7 +13,7 @@ "name": "Claude Haiku 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 35.5 + "terminal-bench-2.0": 13.9 } }, { @@ -21,7 +21,7 @@ "name": "Claude Opus 4.1", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 38.0 + "terminal-bench-2.0": 36.9 } }, { @@ -29,7 +29,7 @@ "name": "Claude Opus 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 59.1 + "terminal-bench-2.0": 51.9 } }, { @@ -37,7 +37,7 @@ "name": "Claude Opus 4.6", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 66.9 + "terminal-bench-2.0": 74.7 } }, { @@ -45,7 +45,7 @@ "name": "Claude Sonnet 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 40.1 + "terminal-bench-2.0": 42.6 } }, { @@ -61,7 +61,7 @@ "name": "Gemini 2.5 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 17.1 + "terminal-bench-2.0": 15.4 } }, { @@ -69,7 +69,7 @@ "name": "Gemini 2.5 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 19.6 + "terminal-bench-2.0": 16.4 } }, { @@ -93,7 +93,7 @@ "name": "Gemini 3.1 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 78.4 + "terminal-bench-2.0": 74.8 } }, { @@ -125,7 +125,7 @@ "name": "Kimi K2 Instruct", "developer": "Moonshot AI", "scores": { - "terminal-bench-2.0": 26.7 + "terminal-bench-2.0": 27.8 } }, { @@ -149,7 +149,7 @@ "name": "Multiple", "developer": "Multiple", "scores": { - "terminal-bench-2.0": 58.4 + "terminal-bench-2.0": 72.4 } }, { @@ -165,7 +165,7 @@ "name": "GPT-5-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 41.3 + "terminal-bench-2.0": 43.4 } }, { @@ -173,7 +173,7 @@ "name": "GPT-5-Mini", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 22.2 + "terminal-bench-2.0": 34.8 } }, { @@ -181,7 +181,7 @@ "name": "GPT-5-Nano", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 7.9 + "terminal-bench-2.0": 7.0 } }, { @@ -197,7 +197,7 @@ "name": "GPT-5.1-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 36.9 + "terminal-bench-2.0": 53.5 } }, { @@ -221,7 +221,7 @@ "name": "GPT-5.2", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 54.0 + "terminal-bench-2.0": 60.7 } }, { @@ -237,7 +237,7 @@ "name": "GPT-5.3-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 77.3 + "terminal-bench-2.0": 74.6 } }, { @@ -253,7 +253,7 @@ "name": "GPT-OSS-20B", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 3.4 + "terminal-bench-2.0": 3.1 } }, { @@ -269,7 +269,7 @@ "name": "Grok Code Fast 1", "developer": "xAI", "scores": { - "terminal-bench-2.0": 14.2 + "terminal-bench-2.0": 25.8 } }, { diff --git a/data/developers/adriszmar.json b/data/developers/adriszmar.json index 1f1d39916960942963a9c3c265196aea3657be38..acb90d745752909d8f96f323acb9caa9b19061ae 100644 --- a/data/developers/adriszmar.json +++ b/data/developers/adriszmar.json @@ -7,12 +7,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1746, - "hfopenllm_v2/BBH": 0.3126, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.245, - "hfopenllm_v2/MUSR": 0.4096, - "hfopenllm_v2/MMLU-PRO": 0.1087 + "hfopenllm_v2/IFEval": 0.1685, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.3963, + "hfopenllm_v2/MMLU-PRO": 0.1066 } } ] diff --git a/data/developers/ai2.json b/data/developers/ai2.json index 4934c11b7806e647da8c3821dcfccba2566bb947..6ae2e91a5d501e1d313c59819f3bee806d5615b0 100644 --- a/data/developers/ai2.json +++ b/data/developers/ai2.json @@ -43,9 +43,9 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7008, - "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3882, + "reward-bench/Score": 0.6924, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.3575, "reward-bench/Safety": 0.7757 } }, diff --git a/data/developers/allenai.json b/data/developers/allenai.json index 26883d607b7a5071bd9da08389772de32edfe871..733bf4fc5c0900a8884a7abb1f0b0f474d01c62a 100644 --- a/data/developers/allenai.json +++ b/data/developers/allenai.json @@ -63,17 +63,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9021, + "reward-bench/Score": 0.7606, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Safety": 0.8844, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8835 } }, { @@ -120,12 +120,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8379, - "hfopenllm_v2/BBH": 0.6157, - "hfopenllm_v2/MATH Level 5": 0.3829, + "hfopenllm_v2/IFEval": 0.8291, + "hfopenllm_v2/BBH": 0.6164, + "hfopenllm_v2/MATH Level 5": 0.4502, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4988, - "hfopenllm_v2/MMLU-PRO": 0.4656 + "hfopenllm_v2/MUSR": 0.4948, + "hfopenllm_v2/MMLU-PRO": 0.4645 } }, { @@ -181,12 +181,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8267, - "hfopenllm_v2/BBH": 0.405, - "hfopenllm_v2/MATH Level 5": 0.1964, - "hfopenllm_v2/GPQA": 0.2987, + "hfopenllm_v2/IFEval": 0.8255, + "hfopenllm_v2/BBH": 0.4061, + "hfopenllm_v2/MATH Level 5": 0.2115, + "hfopenllm_v2/GPQA": 0.297, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2827 + "hfopenllm_v2/MMLU-PRO": 0.2821 } }, { @@ -228,17 +228,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8369, + "reward-bench/Score": 0.6871, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.7588, + "reward-bench/Safety": 0.8644, + "reward-bench/Reasoning": 0.7715, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7642, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.8485, - "reward-bench/Ties": 0.6281, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.7588, - "reward-bench/Reasoning": 0.7715, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6281 } }, { diff --git a/data/developers/amd.json b/data/developers/amd.json index f58c5a6931b8f7bef3756ba18944919a5d99e792..ec85a4364a761e30f1999f359d0d247d8857e139 100644 --- a/data/developers/amd.json +++ b/data/developers/amd.json @@ -7,11 +7,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1918, - "hfopenllm_v2/BBH": 0.2969, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2584, - "hfopenllm_v2/MUSR": 0.3846, + "hfopenllm_v2/IFEval": 0.1842, + "hfopenllm_v2/BBH": 0.2974, + "hfopenllm_v2/MATH Level 5": 0.0053, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.378, "hfopenllm_v2/MMLU-PRO": 0.1169 } } diff --git a/data/developers/anthropic.json b/data/developers/anthropic.json index c1b51c55828f681c4d430c9fc7c3821e0c017451..21c2bc0153b15a2859b6559b8628567638664450 100644 --- a/data/developers/anthropic.json +++ b/data/developers/anthropic.json @@ -436,16 +436,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.5744, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.8378, - "reward-bench/Reasoning": 0.7868, + "reward-bench/Score": 0.8008, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601 + "reward-bench/Ties": 0.5601, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Reasoning": 0.7868 } }, { @@ -525,7 +525,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.5 + "terminal-bench-2.0/terminal-bench-2.0": 13.9 } }, { @@ -650,10 +650,10 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { + "appworld_test_normal/appworld/test_normal": 0.66, "browsecompplus/browsecompplus": 0.49, - "appworld_test_normal/appworld/test_normal": 0.64, - "swe-bench/swe-bench": 0.8072, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, + "swe-bench/swe-bench": 0.65, + "tau-bench-2_airline/tau-bench-2/airline": 0.72, "tau-bench-2_retail/tau-bench-2/retail": 0.85, "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } @@ -664,7 +664,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 38.0 + "terminal-bench-2.0/terminal-bench-2.0": 36.9 } }, { @@ -673,7 +673,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 59.1 + "terminal-bench-2.0/terminal-bench-2.0": 51.9 } }, { @@ -682,7 +682,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 66.9 + "terminal-bench-2.0/terminal-bench-2.0": 74.7 } }, { @@ -756,7 +756,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 40.1 + "terminal-bench-2.0/terminal-bench-2.0": 42.6 } }, { diff --git a/data/developers/arcee-ai.json b/data/developers/arcee-ai.json index fb3c26af6dfe804ff7a40fb7fe0b6065ba67af15..a67ed73a3aff7f3d10c24e6ce76a817fd626a3c8 100644 --- a/data/developers/arcee-ai.json +++ b/data/developers/arcee-ai.json @@ -49,12 +49,12 @@ "developer": "arcee-ai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5621, - "hfopenllm_v2/BBH": 0.5489, - "hfopenllm_v2/MATH Level 5": 0.2953, - "hfopenllm_v2/GPQA": 0.307, - "hfopenllm_v2/MUSR": 0.4021, - "hfopenllm_v2/MMLU-PRO": 0.3822 + "hfopenllm_v2/IFEval": 0.5718, + "hfopenllm_v2/BBH": 0.5481, + "hfopenllm_v2/MATH Level 5": 0.114, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4008, + "hfopenllm_v2/MMLU-PRO": 0.3813 } }, { diff --git a/data/developers/atanddev.json b/data/developers/atanddev.json index d269c1fbba37e9ea4df30635a880656764997806..42530fda0188ea30162bbff58f1010442b0394c9 100644 --- a/data/developers/atanddev.json +++ b/data/developers/atanddev.json @@ -7,12 +7,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4511, - "hfopenllm_v2/BBH": 0.4275, - "hfopenllm_v2/MATH Level 5": 0.1473, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3623, - "hfopenllm_v2/MMLU-PRO": 0.2806 + "hfopenllm_v2/IFEval": 0.4605, + "hfopenllm_v2/BBH": 0.4258, + "hfopenllm_v2/MATH Level 5": 0.0748, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3636, + "hfopenllm_v2/MMLU-PRO": 0.2812 } } ] diff --git a/data/developers/boltmonkey.json b/data/developers/boltmonkey.json index d93c347b80733b690b666d77f058a8e36708eee2..93a1093c0cbb01b9a84ae22cf6063fc50d8793cb 100644 --- a/data/developers/boltmonkey.json +++ b/data/developers/boltmonkey.json @@ -21,12 +21,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.459, - "hfopenllm_v2/BBH": 0.5185, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.2743, - "hfopenllm_v2/MUSR": 0.4083, - "hfopenllm_v2/MMLU-PRO": 0.3631 + "hfopenllm_v2/IFEval": 0.7999, + "hfopenllm_v2/BBH": 0.5152, + "hfopenllm_v2/MATH Level 5": 0.1193, + "hfopenllm_v2/GPQA": 0.281, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3733 } }, { diff --git a/data/developers/cognitivecomputations.json b/data/developers/cognitivecomputations.json index 292d5e513d244c8ebb441a809017ff16919c9354..27ef3ede420acfeab83ed5bb754062e32374c41a 100644 --- a/data/developers/cognitivecomputations.json +++ b/data/developers/cognitivecomputations.json @@ -77,12 +77,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3613, - "hfopenllm_v2/BBH": 0.6123, - "hfopenllm_v2/MATH Level 5": 0.1239, - "hfopenllm_v2/GPQA": 0.328, - "hfopenllm_v2/MUSR": 0.4112, - "hfopenllm_v2/MMLU-PRO": 0.4494 + "hfopenllm_v2/IFEval": 0.4124, + "hfopenllm_v2/BBH": 0.6383, + "hfopenllm_v2/MATH Level 5": 0.182, + "hfopenllm_v2/GPQA": 0.3289, + "hfopenllm_v2/MUSR": 0.4349, + "hfopenllm_v2/MMLU-PRO": 0.4525 } }, { diff --git a/data/developers/columbia-nlp.json b/data/developers/columbia-nlp.json index b04d1f97bca6939c03a53d86cd396214edf82f72..11f5fed45eb39522787aef4e964f3d3e28d320c0 100644 --- a/data/developers/columbia-nlp.json +++ b/data/developers/columbia-nlp.json @@ -7,12 +7,12 @@ "developer": "Columbia-NLP", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3278, - "hfopenllm_v2/BBH": 0.392, - "hfopenllm_v2/MATH Level 5": 0.0431, - "hfopenllm_v2/GPQA": 0.2492, - "hfopenllm_v2/MUSR": 0.412, - "hfopenllm_v2/MMLU-PRO": 0.1666 + "hfopenllm_v2/IFEval": 0.3102, + "hfopenllm_v2/BBH": 0.3881, + "hfopenllm_v2/MATH Level 5": 0.0536, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.4081, + "hfopenllm_v2/MMLU-PRO": 0.1665 } }, { diff --git a/data/developers/cpayne1303.json b/data/developers/cpayne1303.json index 6d735bd94a67b9fc86d407e5a74d4ec119a21a01..878ab50b2c306395a2d661382c84d0660b0f0d51 100644 --- a/data/developers/cpayne1303.json +++ b/data/developers/cpayne1303.json @@ -35,12 +35,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1916, - "hfopenllm_v2/BBH": 0.2977, - "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/IFEval": 0.1949, + "hfopenllm_v2/BBH": 0.2965, + "hfopenllm_v2/MATH Level 5": 0.0045, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3872, - "hfopenllm_v2/MMLU-PRO": 0.1132 + "hfopenllm_v2/MUSR": 0.3885, + "hfopenllm_v2/MMLU-PRO": 0.1111 } }, { diff --git a/data/developers/daemontatox.json b/data/developers/daemontatox.json index 3de1c87bc255ec29e11a7fcd9434ebc4d17ff27a..b8e6a28a7fbf1f38ce6fe03abcdad2859f985a0e 100644 --- a/data/developers/daemontatox.json +++ b/data/developers/daemontatox.json @@ -119,12 +119,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.777, - "hfopenllm_v2/BBH": 0.5187, - "hfopenllm_v2/MATH Level 5": 0.2198, - "hfopenllm_v2/GPQA": 0.2936, - "hfopenllm_v2/MUSR": 0.3911, - "hfopenllm_v2/MMLU-PRO": 0.3738 + "hfopenllm_v2/IFEval": 0.5064, + "hfopenllm_v2/BBH": 0.5112, + "hfopenllm_v2/MATH Level 5": 0.1631, + "hfopenllm_v2/GPQA": 0.3163, + "hfopenllm_v2/MUSR": 0.3973, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { @@ -231,12 +231,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4855, - "hfopenllm_v2/BBH": 0.6627, - "hfopenllm_v2/MATH Level 5": 0.4841, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4256, - "hfopenllm_v2/MMLU-PRO": 0.5542 + "hfopenllm_v2/IFEval": 0.3745, + "hfopenllm_v2/BBH": 0.6668, + "hfopenllm_v2/MATH Level 5": 0.4758, + "hfopenllm_v2/GPQA": 0.3943, + "hfopenllm_v2/MUSR": 0.4858, + "hfopenllm_v2/MMLU-PRO": 0.5593 } }, { diff --git a/data/developers/deepmount00.json b/data/developers/deepmount00.json index 5505c28134c7ac10ef3a27978dd2745253c793a1..e898074e4a61782e05002f7b47eb2ee0411313aa 100644 --- a/data/developers/deepmount00.json +++ b/data/developers/deepmount00.json @@ -63,12 +63,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7917, - "hfopenllm_v2/BBH": 0.5109, - "hfopenllm_v2/MATH Level 5": 0.1088, - "hfopenllm_v2/GPQA": 0.2878, - "hfopenllm_v2/MUSR": 0.4136, - "hfopenllm_v2/MMLU-PRO": 0.3876 + "hfopenllm_v2/IFEval": 0.5365, + "hfopenllm_v2/BBH": 0.517, + "hfopenllm_v2/MATH Level 5": 0.1707, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4487, + "hfopenllm_v2/MMLU-PRO": 0.396 } }, { diff --git a/data/developers/doppelreflex.json b/data/developers/doppelreflex.json index 4e478be7761570eec41b8a497db82c0ab081c8b0..0b78fc1cbec001d1df6b60b1fab265b2eab6799e 100644 --- a/data/developers/doppelreflex.json +++ b/data/developers/doppelreflex.json @@ -175,12 +175,12 @@ "developer": "DoppelReflEx", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.451, - "hfopenllm_v2/BBH": 0.4944, - "hfopenllm_v2/MATH Level 5": 0.1156, - "hfopenllm_v2/GPQA": 0.3196, - "hfopenllm_v2/MUSR": 0.3896, - "hfopenllm_v2/MMLU-PRO": 0.3256 + "hfopenllm_v2/IFEval": 0.436, + "hfopenllm_v2/BBH": 0.4956, + "hfopenllm_v2/MATH Level 5": 0.0589, + "hfopenllm_v2/GPQA": 0.3205, + "hfopenllm_v2/MUSR": 0.3843, + "hfopenllm_v2/MMLU-PRO": 0.3237 } }, { diff --git a/data/developers/epistemeai.json b/data/developers/epistemeai.json index 11b25cd31eead1ba78e7fde536703493137e208b..ae59684b2ae8edda61f6033386c80ae35c1570fc 100644 --- a/data/developers/epistemeai.json +++ b/data/developers/epistemeai.json @@ -231,12 +231,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7207, - "hfopenllm_v2/BBH": 0.461, - "hfopenllm_v2/MATH Level 5": 0.1314, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3432, - "hfopenllm_v2/MMLU-PRO": 0.3354 + "hfopenllm_v2/IFEval": 0.7305, + "hfopenllm_v2/BBH": 0.4649, + "hfopenllm_v2/MATH Level 5": 0.1397, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3209, + "hfopenllm_v2/MMLU-PRO": 0.348 } }, { diff --git a/data/developers/etherll.json b/data/developers/etherll.json index 2be2455f8558b72fcbd342c72cc671c443f3155e..6a72dd37a279f4d76a8244a057714c910854bcd2 100644 --- a/data/developers/etherll.json +++ b/data/developers/etherll.json @@ -35,12 +35,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4672, - "hfopenllm_v2/BBH": 0.5013, - "hfopenllm_v2/MATH Level 5": 0.0279, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.386, - "hfopenllm_v2/MMLU-PRO": 0.3482 + "hfopenllm_v2/IFEval": 0.6106, + "hfopenllm_v2/BBH": 0.5347, + "hfopenllm_v2/MATH Level 5": 0.1548, + "hfopenllm_v2/GPQA": 0.3146, + "hfopenllm_v2/MUSR": 0.3991, + "hfopenllm_v2/MMLU-PRO": 0.3752 } }, { diff --git a/data/developers/fblgit.json b/data/developers/fblgit.json index 6fc1e61b7aa2564e949656f946971c56834a0ca1..c9c68512a8d17a7002851c775ec2736dba6e4b4f 100644 --- a/data/developers/fblgit.json +++ b/data/developers/fblgit.json @@ -91,12 +91,12 @@ "developer": "fblgit", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5181, - "hfopenllm_v2/BBH": 0.7033, - "hfopenllm_v2/MATH Level 5": 0.4947, - "hfopenllm_v2/GPQA": 0.3826, - "hfopenllm_v2/MUSR": 0.5008, - "hfopenllm_v2/MMLU-PRO": 0.5915 + "hfopenllm_v2/IFEval": 0.4503, + "hfopenllm_v2/BBH": 0.7035, + "hfopenllm_v2/MATH Level 5": 0.3943, + "hfopenllm_v2/GPQA": 0.401, + "hfopenllm_v2/MUSR": 0.5021, + "hfopenllm_v2/MMLU-PRO": 0.5911 } }, { diff --git a/data/developers/goekdeniz-guelmez.json b/data/developers/goekdeniz-guelmez.json index e3a743ff451afc2206e344fc708fd30240018483..c6e66fe9b3a75e2d49b70b9c90b10752326b37c4 100644 --- a/data/developers/goekdeniz-guelmez.json +++ b/data/developers/goekdeniz-guelmez.json @@ -49,12 +49,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7598, - "hfopenllm_v2/BBH": 0.5107, - "hfopenllm_v2/MATH Level 5": 0.4237, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.4539, - "hfopenllm_v2/MMLU-PRO": 0.4012 + "hfopenllm_v2/IFEval": 0.7628, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2802, + "hfopenllm_v2/MUSR": 0.4579, + "hfopenllm_v2/MMLU-PRO": 0.4033 } }, { diff --git a/data/developers/google.json b/data/developers/google.json index ae9dc7717f1b79909d67dfe1364f96967d93b736..81f1df0599e79d5fa5b9eb113ec92dededc4447e 100644 --- a/data/developers/google.json +++ b/data/developers/google.json @@ -139,6 +139,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -146,7 +147,6 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, - "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -723,7 +723,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 17.1 + "terminal-bench-2.0/terminal-bench-2.0": 15.4 } }, { @@ -823,7 +823,7 @@ "reward-bench/Safety": 0.881, "reward-bench/Focus": 0.805, "reward-bench/Ties": 0.811, - "terminal-bench-2.0/terminal-bench-2.0": 19.6 + "terminal-bench-2.0/terminal-bench-2.0": 16.4 } }, { @@ -879,7 +879,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.13, + "appworld_test_normal/appworld/test_normal": 0.36, "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, @@ -901,9 +901,9 @@ "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.7, - "tau-bench-2_retail/tau-bench-2/retail": 0.7805, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 + "tau-bench-2_airline/tau-bench-2/airline": 0.68, + "tau-bench-2_retail/tau-bench-2/retail": 0.82, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 } }, { @@ -912,7 +912,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 78.4 + "terminal-bench-2.0/terminal-bench-2.0": 74.8 } }, { diff --git a/data/developers/gunulhona.json b/data/developers/gunulhona.json index 1eba4dc6aed35ea93d90c3c8c2e1a2676805ffa0..3d63c85d80df83c6632e31f815d8dc78510d19a0 100644 --- a/data/developers/gunulhona.json +++ b/data/developers/gunulhona.json @@ -21,12 +21,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4441, - "hfopenllm_v2/BBH": 0.4863, + "hfopenllm_v2/IFEval": 0.288, + "hfopenllm_v2/BBH": 0.5154, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.307, - "hfopenllm_v2/MUSR": 0.3986, - "hfopenllm_v2/MMLU-PRO": 0.3098 + "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/MUSR": 0.408, + "hfopenllm_v2/MMLU-PRO": 0.3817 } } ] diff --git a/data/developers/hendrydong.json b/data/developers/hendrydong.json index 56e05071d3b06a1289763aa84d052b49b755b3aa..66d1cb6cae0c438dba7144b9a92c7891857c219e 100644 --- a/data/developers/hendrydong.json +++ b/data/developers/hendrydong.json @@ -7,17 +7,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7847, + "reward-bench/Score": 0.5851, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Safety": 0.6956, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, - "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508 + "reward-bench/Ties": 0.5988 } } ] diff --git a/data/developers/huggingfacetb.json b/data/developers/huggingfacetb.json index 60e6e856fd10ebf29a5d127dd18301cd9edf10dd..bed31781473fb30427be579aab45ef01bf5054ce 100644 --- a/data/developers/huggingfacetb.json +++ b/data/developers/huggingfacetb.json @@ -161,12 +161,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3842, - "hfopenllm_v2/BBH": 0.3144, - "hfopenllm_v2/MATH Level 5": 0.0151, - "hfopenllm_v2/GPQA": 0.255, - "hfopenllm_v2/MUSR": 0.3461, - "hfopenllm_v2/MMLU-PRO": 0.1117 + "hfopenllm_v2/IFEval": 0.083, + "hfopenllm_v2/BBH": 0.3053, + "hfopenllm_v2/MATH Level 5": 0.0083, + "hfopenllm_v2/GPQA": 0.2651, + "hfopenllm_v2/MUSR": 0.3423, + "hfopenllm_v2/MMLU-PRO": 0.1126 } } ] diff --git a/data/developers/infly.json b/data/developers/infly.json index fe3f0dc6f7a4b2c08dd2895544fd05de4f16df3c..d497bf1e2632542284e99f21127cba81c8ed1b97 100644 --- a/data/developers/infly.json +++ b/data/developers/infly.json @@ -7,16 +7,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7648, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Safety": 0.9644, - "reward-bench/Reasoning": 0.9912, + "reward-bench/Score": 0.9511, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622 + "reward-bench/Ties": 0.8622, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Reasoning": 0.9912 } } ] diff --git a/data/developers/internlm.json b/data/developers/internlm.json index ba5efe87a4d9e153aadeeeaadd0b465b5955bdd6..035007101b6dacaf9414107420239eb159a0ec86 100644 --- a/data/developers/internlm.json +++ b/data/developers/internlm.json @@ -21,16 +21,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -39,16 +39,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9016, + "reward-bench/Score": 0.5628, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Safety": 0.6111, + "reward-bench/Reasoning": 0.9576, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, - "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Reasoning": 0.9576 + "reward-bench/Ties": 0.5483 } }, { diff --git a/data/developers/isaak-carter.json b/data/developers/isaak-carter.json index 690871bc9bb32834a871ee3cfe82d0488b7c6bad..5e4243cedab9b62fa1fd1f8e9be8ea7f6c708259 100644 --- a/data/developers/isaak-carter.json +++ b/data/developers/isaak-carter.json @@ -35,12 +35,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2477, - "hfopenllm_v2/BBH": 0.4758, - "hfopenllm_v2/MATH Level 5": 0.0453, - "hfopenllm_v2/GPQA": 0.2911, - "hfopenllm_v2/MUSR": 0.3641, - "hfopenllm_v2/MMLU-PRO": 0.3292 + "hfopenllm_v2/IFEval": 0.2553, + "hfopenllm_v2/BBH": 0.4725, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.2919, + "hfopenllm_v2/MUSR": 0.3654, + "hfopenllm_v2/MMLU-PRO": 0.3316 } } ] diff --git a/data/developers/jaspionjader.json b/data/developers/jaspionjader.json index 053d128582b4aa02040ae51cd0577288669f17e5..9d9d1e268e56a9945ae657deca0493de6a22ce3d 100644 --- a/data/developers/jaspionjader.json +++ b/data/developers/jaspionjader.json @@ -1477,12 +1477,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4345, - "hfopenllm_v2/BBH": 0.5419, - "hfopenllm_v2/MATH Level 5": 0.1292, - "hfopenllm_v2/GPQA": 0.3087, + "hfopenllm_v2/IFEval": 0.4418, + "hfopenllm_v2/BBH": 0.5406, + "hfopenllm_v2/MATH Level 5": 0.1352, + "hfopenllm_v2/GPQA": 0.3062, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.3854 + "hfopenllm_v2/MMLU-PRO": 0.386 } }, { diff --git a/data/developers/kavonalds.json b/data/developers/kavonalds.json index c5a641658de284c1f8e0579f706e8c77674e2dfa..3c38b58473bebadc89fb67cf205ea04b79c73205 100644 --- a/data/developers/kavonalds.json +++ b/data/developers/kavonalds.json @@ -7,12 +7,12 @@ "developer": "kavonalds", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2701, - "hfopenllm_v2/BBH": 0.5566, + "hfopenllm_v2/IFEval": 0.3283, + "hfopenllm_v2/BBH": 0.6651, "hfopenllm_v2/MATH Level 5": 0.068, - "hfopenllm_v2/GPQA": 0.2802, - "hfopenllm_v2/MUSR": 0.3682, - "hfopenllm_v2/MMLU-PRO": 0.1449 + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3393, + "hfopenllm_v2/MMLU-PRO": 0.1314 } }, { diff --git a/data/developers/lxzgordon.json b/data/developers/lxzgordon.json index e4ace3cc8f7193c8c711403c980535ee73fdd6d3..7f802cf733857054e01537f3ecf745a3fdb38a05 100644 --- a/data/developers/lxzgordon.json +++ b/data/developers/lxzgordon.json @@ -20,16 +20,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9294, + "reward-bench/Score": 0.7394, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Safety": 0.9178, + "reward-bench/Reasoning": 0.9698, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Reasoning": 0.9698 + "reward-bench/Ties": 0.7653 } } ] diff --git a/data/developers/magpie-align.json b/data/developers/magpie-align.json index 155a416e715db19186e6af681f6736ea9ed101d1..0dc0a43e89bb456a61006caa30051add55effb08 100644 --- a/data/developers/magpie-align.json +++ b/data/developers/magpie-align.json @@ -35,12 +35,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4118, - "hfopenllm_v2/BBH": 0.4811, - "hfopenllm_v2/MATH Level 5": 0.034, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3047, - "hfopenllm_v2/MMLU-PRO": 0.3006 + "hfopenllm_v2/IFEval": 0.4027, + "hfopenllm_v2/BBH": 0.4789, + "hfopenllm_v2/MATH Level 5": 0.0461, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.3087, + "hfopenllm_v2/MMLU-PRO": 0.3001 } }, { diff --git a/data/developers/meta-llama.json b/data/developers/meta-llama.json index a28b7c096e670398e8b74d7b30002bc56044221b..76d923e16aab37465ed7dda94d7be71e39f4b29e 100644 --- a/data/developers/meta-llama.json +++ b/data/developers/meta-llama.json @@ -265,12 +265,12 @@ "developer": "meta-llama", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4782, - "hfopenllm_v2/BBH": 0.491, - "hfopenllm_v2/MATH Level 5": 0.0914, - "hfopenllm_v2/GPQA": 0.2928, - "hfopenllm_v2/MUSR": 0.3805, - "hfopenllm_v2/MMLU-PRO": 0.3591, + "hfopenllm_v2/IFEval": 0.7408, + "hfopenllm_v2/BBH": 0.4989, + "hfopenllm_v2/MATH Level 5": 0.0869, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3568, + "hfopenllm_v2/MMLU-PRO": 0.3664, "reward-bench/Score": 0.645, "reward-bench/Chat": 0.8547, "reward-bench/Chat Hard": 0.4156, diff --git a/data/developers/mistralai.json b/data/developers/mistralai.json index 57d196d824fe3b2d3e0a8303970991b9f16872ee..168bb98ff1b313fc7d40f024df899fec3a02671f 100644 --- a/data/developers/mistralai.json +++ b/data/developers/mistralai.json @@ -513,12 +513,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6283, - "hfopenllm_v2/BBH": 0.583, - "hfopenllm_v2/MATH Level 5": 0.2039, - "hfopenllm_v2/GPQA": 0.3331, - "hfopenllm_v2/MUSR": 0.4063, - "hfopenllm_v2/MMLU-PRO": 0.4099 + "hfopenllm_v2/IFEval": 0.667, + "hfopenllm_v2/BBH": 0.5213, + "hfopenllm_v2/MATH Level 5": 0.1435, + "hfopenllm_v2/GPQA": 0.3238, + "hfopenllm_v2/MUSR": 0.3632, + "hfopenllm_v2/MMLU-PRO": 0.396 } }, { diff --git a/data/developers/mlabonne.json b/data/developers/mlabonne.json index 2620a8c4e8931697abdcd44e4a4aae7c1e430da5..be86bd7fe732025f133b06dc7c412aa5e56b7119 100644 --- a/data/developers/mlabonne.json +++ b/data/developers/mlabonne.json @@ -161,12 +161,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7561, - "hfopenllm_v2/BBH": 0.5111, - "hfopenllm_v2/MATH Level 5": 0.0906, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3841 + "hfopenllm_v2/IFEval": 0.4162, + "hfopenllm_v2/BBH": 0.5124, + "hfopenllm_v2/MATH Level 5": 0.0853, + "hfopenllm_v2/GPQA": 0.3029, + "hfopenllm_v2/MUSR": 0.415, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { diff --git a/data/developers/moonshot_ai.json b/data/developers/moonshot_ai.json index d83f11402fcca0c39c33d09eed495f2aefd69384..746185ce773a539bd025922ab856fdcf2f8a1d9f 100644 --- a/data/developers/moonshot_ai.json +++ b/data/developers/moonshot_ai.json @@ -7,7 +7,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 26.7 + "terminal-bench-2.0/terminal-bench-2.0": 27.8 } }, { diff --git a/data/developers/multiple.json b/data/developers/multiple.json index 3ca416f37a8a2d37629452327cce332c02e2b948..34cdb844d495e12fd3a3820204fbda313306e211 100644 --- a/data/developers/multiple.json +++ b/data/developers/multiple.json @@ -7,7 +7,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 58.4 + "terminal-bench-2.0/terminal-bench-2.0": 72.4 } } ] diff --git a/data/developers/ncsoft.json b/data/developers/ncsoft.json index 78886a7e33419ae72417f31ea13b0a41a2a3ced1..5cc19c52d2bf4d85f74e73a3056f8da048c42869 100644 --- a/data/developers/ncsoft.json +++ b/data/developers/ncsoft.json @@ -20,16 +20,16 @@ "developer": "NCSOFT", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8942, + "reward-bench/Score": 0.648, + "reward-bench/Chat": 0.9721, + "reward-bench/Chat Hard": 0.818, + "reward-bench/Safety": 0.7222, + "reward-bench/Reasoning": 0.9192, "reward-bench/Factuality": 0.6084, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.9596, - "reward-bench/Ties": 0.6786, - "reward-bench/Chat": 0.9721, - "reward-bench/Chat Hard": 0.818, - "reward-bench/Reasoning": 0.9192 + "reward-bench/Ties": 0.6786 } }, { diff --git a/data/developers/nexusflow.json b/data/developers/nexusflow.json index 49fe739e7104f59b148ecb24e067de8a0cb500b5..2f78cab8780e463261ce35ef251c295eb3f3fd0f 100644 --- a/data/developers/nexusflow.json +++ b/data/developers/nexusflow.json @@ -21,17 +21,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4553, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Safety": 0.7556, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137, + "reward-bench/Score": 0.8133, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, + "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004 + "reward-bench/Ties": 0.1004, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137 } } ] diff --git a/data/developers/ontocord.json b/data/developers/ontocord.json index add41dbe183800ebed90c916e0b704c974dd50e7..c16f4cbacdda3485b721f459b079923a6793a670 100644 --- a/data/developers/ontocord.json +++ b/data/developers/ontocord.json @@ -273,12 +273,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { diff --git a/data/developers/openai.json b/data/developers/openai.json index ede5404bc51db07d8e63c7bbcb6be1b005184ac0..778495ca5d9933e50afc2a386628c3370e3c9523 100644 --- a/data/developers/openai.json +++ b/data/developers/openai.json @@ -163,16 +163,16 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { + "ace/Overall Score": 0.515, + "ace/Food Score": 0.65, + "ace/Gaming Score": 0.578, "apex-agents/Overall Pass@1": 0.23, "apex-agents/Overall Pass@8": 0.4, "apex-agents/Overall Mean Score": 0.387, "apex-agents/Investment Banking Pass@1": 0.273, "apex-agents/Management Consulting Pass@1": 0.227, "apex-agents/Corporate Law Pass@1": 0.189, - "apex-agents/Corporate Lawyer Mean Score": 0.443, - "ace/Overall Score": 0.515, - "ace/Food Score": 0.65, - "ace/Gaming Score": 0.578 + "apex-agents/Corporate Lawyer Mean Score": 0.443 } }, { @@ -772,16 +772,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.8673, + "reward-bench/Score": 0.6493, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.8619, + "reward-bench/Reasoning": 0.8661, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, - "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.8661 + "reward-bench/Ties": 0.7819 } }, { @@ -859,16 +859,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 } }, { @@ -922,7 +922,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 41.3 + "terminal-bench-2.0/terminal-bench-2.0": 43.4 } }, { @@ -931,7 +931,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 22.2 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -954,7 +954,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.9 + "terminal-bench-2.0/terminal-bench-2.0": 7.0 } }, { @@ -986,7 +986,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 + "terminal-bench-2.0/terminal-bench-2.0": 53.5 } }, { @@ -1013,7 +1013,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 54.0 + "terminal-bench-2.0/terminal-bench-2.0": 60.7 } }, { @@ -1022,15 +1022,15 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.48, + "appworld_test_normal/appworld/test_normal": 0.071, + "browsecompplus/browsecompplus": 0.46, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.5455, - "tau-bench-2_airline/tau-bench-2/airline": 0.48, - "tau-bench-2_retail/tau-bench-2/retail": 0.51, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 + "swe-bench/swe-bench": 0.57, + "tau-bench-2_airline/tau-bench-2/airline": 0.6, + "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 } }, { @@ -1048,7 +1048,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 77.3 + "terminal-bench-2.0/terminal-bench-2.0": 74.6 } }, { @@ -1130,7 +1130,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.056338028169014086, "livecodebenchpro/Easy Problems": 0.5070422535211268, - "terminal-bench-2.0/terminal-bench-2.0": 3.4 + "terminal-bench-2.0/terminal-bench-2.0": 3.1 } }, { diff --git a/data/developers/openassistant.json b/data/developers/openassistant.json index 95e3fbf7702e838e1924ccdd51ef3a7ea27c9d5c..94a84f4efb74c56aff95d97f1dc674cfa0445541 100644 --- a/data/developers/openassistant.json +++ b/data/developers/openassistant.json @@ -59,17 +59,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.32, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Safety": 0.3667, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836, + "reward-bench/Score": 0.6126, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, + "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12 + "reward-bench/Ties": 0.12, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836 } } ] diff --git a/data/developers/openbmb.json b/data/developers/openbmb.json index b9795ed819d46987af7a443e489394fa08308cdd..d8dae84054074ce01b5c47fc58b69a148fdc99c0 100644 --- a/data/developers/openbmb.json +++ b/data/developers/openbmb.json @@ -21,17 +21,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5806, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Safety": 0.6267, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172, + "reward-bench/Score": 0.8159, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972 + "reward-bench/Ties": 0.5972, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172 } }, { diff --git a/data/developers/pku-alignment.json b/data/developers/pku-alignment.json index 4cebe3d4c55c06b2bbbd44b7bb6fe487a030b922..b42d9be156feba13e7cdbffbd62351a64e28c1fb 100644 --- a/data/developers/pku-alignment.json +++ b/data/developers/pku-alignment.json @@ -7,17 +7,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3332, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Safety": 0.7589, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57, + "reward-bench/Score": 0.5798, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, + "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57 } }, { diff --git a/data/developers/prithivmlmods.json b/data/developers/prithivmlmods.json index 88e743a4279f310dec935aed3968189a78be084e..f14f3864a910968b11c756eb8300bd4bd3eea36e 100644 --- a/data/developers/prithivmlmods.json +++ b/data/developers/prithivmlmods.json @@ -63,12 +63,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6052, - "hfopenllm_v2/BBH": 0.6317, - "hfopenllm_v2/MATH Level 5": 0.4789, - "hfopenllm_v2/GPQA": 0.3742, - "hfopenllm_v2/MUSR": 0.486, - "hfopenllm_v2/MMLU-PRO": 0.5302 + "hfopenllm_v2/IFEval": 0.6064, + "hfopenllm_v2/BBH": 0.6296, + "hfopenllm_v2/MATH Level 5": 0.3708, + "hfopenllm_v2/GPQA": 0.3733, + "hfopenllm_v2/MUSR": 0.4873, + "hfopenllm_v2/MMLU-PRO": 0.5307 } }, { diff --git a/data/developers/qingy2019.json b/data/developers/qingy2019.json index b11a9d79a541432fdd0c8e33c56cfaa201540388..3885f54f7f780ebe87cee5b8aacc1c1136f4441f 100644 --- a/data/developers/qingy2019.json +++ b/data/developers/qingy2019.json @@ -49,12 +49,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { diff --git a/data/developers/ray2333.json b/data/developers/ray2333.json index 51f9b51d9ebb192a98a4a9a46633e05cf73f77b6..709d24161c8237750555868d00eabe667376cddb 100644 --- a/data/developers/ray2333.json +++ b/data/developers/ray2333.json @@ -79,17 +79,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8464, + "reward-bench/Score": 0.589, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Safety": 0.7222, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209 + "reward-bench/Ties": 0.5743 } }, { @@ -98,16 +98,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9154, + "reward-bench/Score": 0.6766, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8618, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9362, "reward-bench/Factuality": 0.6274, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.5847, - "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.8929, - "reward-bench/Ties": 0.6824, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8618, - "reward-bench/Reasoning": 0.9362 + "reward-bench/Ties": 0.6824 } }, { @@ -116,17 +116,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8542, + "reward-bench/Score": 0.6089, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Safety": 0.7867, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309 + "reward-bench/Ties": 0.5981 } }, { diff --git a/data/developers/recoilme.json b/data/developers/recoilme.json index 50baa7f16a67b3ee26a355a9f5b41de814bc7e52..a767d74249450554babea0e0fb1705321ef571fe 100644 --- a/data/developers/recoilme.json +++ b/data/developers/recoilme.json @@ -7,12 +7,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2854, - "hfopenllm_v2/BBH": 0.5984, - "hfopenllm_v2/MATH Level 5": 0.1005, - "hfopenllm_v2/GPQA": 0.3297, - "hfopenllm_v2/MUSR": 0.4607, - "hfopenllm_v2/MMLU-PRO": 0.4162 + "hfopenllm_v2/IFEval": 0.7649, + "hfopenllm_v2/BBH": 0.5974, + "hfopenllm_v2/MATH Level 5": 0.0174, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4245, + "hfopenllm_v2/MMLU-PRO": 0.4207 } }, { @@ -35,12 +35,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2747, - "hfopenllm_v2/BBH": 0.6031, - "hfopenllm_v2/MATH Level 5": 0.0831, - "hfopenllm_v2/GPQA": 0.3305, - "hfopenllm_v2/MUSR": 0.4686, - "hfopenllm_v2/MMLU-PRO": 0.4122 + "hfopenllm_v2/IFEval": 0.7592, + "hfopenllm_v2/BBH": 0.6026, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.3289, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.4163 } }, { @@ -49,12 +49,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7439, - "hfopenllm_v2/BBH": 0.5993, - "hfopenllm_v2/MATH Level 5": 0.0876, - "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4204, - "hfopenllm_v2/MMLU-PRO": 0.4072 + "hfopenllm_v2/IFEval": 0.5761, + "hfopenllm_v2/BBH": 0.602, + "hfopenllm_v2/MATH Level 5": 0.1888, + "hfopenllm_v2/GPQA": 0.3372, + "hfopenllm_v2/MUSR": 0.4632, + "hfopenllm_v2/MMLU-PRO": 0.4039 } }, { diff --git a/data/developers/replete-ai.json b/data/developers/replete-ai.json index dbb06f00736a7fcddf76b24a5c7673e098ecab57..0f038b31b0c0cf28a02b6c25fb4fd9bd374c118c 100644 --- a/data/developers/replete-ai.json +++ b/data/developers/replete-ai.json @@ -91,12 +91,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0932, - "hfopenllm_v2/BBH": 0.2977, + "hfopenllm_v2/IFEval": 0.0905, + "hfopenllm_v2/BBH": 0.2985, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2475, - "hfopenllm_v2/MUSR": 0.3941, - "hfopenllm_v2/MMLU-PRO": 0.1157 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3848, + "hfopenllm_v2/MMLU-PRO": 0.1158 } }, { diff --git a/data/developers/riaz.json b/data/developers/riaz.json index d54f5bbaf70d322fa875b56275f987dd3ae70f34..342d5654379425c9866753c909eb0413c18e5c77 100644 --- a/data/developers/riaz.json +++ b/data/developers/riaz.json @@ -7,12 +7,12 @@ "developer": "riaz", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4373, - "hfopenllm_v2/BBH": 0.4586, - "hfopenllm_v2/MATH Level 5": 0.0514, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3763, - "hfopenllm_v2/MMLU-PRO": 0.2964 + "hfopenllm_v2/IFEval": 0.4137, + "hfopenllm_v2/BBH": 0.4565, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.276, + "hfopenllm_v2/MUSR": 0.3776, + "hfopenllm_v2/MMLU-PRO": 0.2978 } } ] diff --git a/data/developers/rombodawg.json b/data/developers/rombodawg.json index e7b347a91d5be7f4c52661e255a50c99caa3a3fd..a241fbd73dcecc7fc768ef47719ff98c1bcb64ab 100644 --- a/data/developers/rombodawg.json +++ b/data/developers/rombodawg.json @@ -133,12 +133,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2566, - "hfopenllm_v2/BBH": 0.39, - "hfopenllm_v2/MATH Level 5": 0.1208, - "hfopenllm_v2/GPQA": 0.2626, + "hfopenllm_v2/IFEval": 0.2595, + "hfopenllm_v2/BBH": 0.3884, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2743, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2741 + "hfopenllm_v2/MMLU-PRO": 0.2719 } }, { diff --git a/data/developers/sao10k.json b/data/developers/sao10k.json index 6569058a39b468de833ea3ab9419e073af211e5a..e66900f1906266fdf37e55995d3b04130f5b6fd2 100644 --- a/data/developers/sao10k.json +++ b/data/developers/sao10k.json @@ -35,12 +35,12 @@ "developer": "Sao10K", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7281, - "hfopenllm_v2/BBH": 0.6503, - "hfopenllm_v2/MATH Level 5": 0.2243, + "hfopenllm_v2/IFEval": 0.7384, + "hfopenllm_v2/BBH": 0.6471, + "hfopenllm_v2/MATH Level 5": 0.2137, "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4196, - "hfopenllm_v2/MMLU-PRO": 0.5096 + "hfopenllm_v2/MUSR": 0.4209, + "hfopenllm_v2/MMLU-PRO": 0.5104 } }, { diff --git a/data/developers/shikaichen.json b/data/developers/shikaichen.json index 6162cba6dec7eaed27af88272ffb98342af2522b..6502ef5b0e02cc85862537458f25816eb0826b7d 100644 --- a/data/developers/shikaichen.json +++ b/data/developers/shikaichen.json @@ -7,16 +7,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } } ] diff --git a/data/developers/skywork.json b/data/developers/skywork.json index a1886dbde5ab2f08ceeb175a3c92e5b11655f24d..b75ac4fa3e86dd1c341887c14306e39e694944d2 100644 --- a/data/developers/skywork.json +++ b/data/developers/skywork.json @@ -71,16 +71,16 @@ "hfopenllm_v2/GPQA": 0.344, "hfopenllm_v2/MUSR": 0.4231, "hfopenllm_v2/MMLU-PRO": 0.4103, - "reward-bench/Score": 0.7531, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.8991, - "reward-bench/Safety": 0.9689, - "reward-bench/Reasoning": 0.9807, + "reward-bench/Score": 0.9426, "reward-bench/Factuality": 0.7674, "reward-bench/Precise IF": 0.375, "reward-bench/Math": 0.6721, + "reward-bench/Safety": 0.9297, "reward-bench/Focus": 0.9172, - "reward-bench/Ties": 0.8182 + "reward-bench/Ties": 0.8182, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.8991, + "reward-bench/Reasoning": 0.9807 } }, { diff --git a/data/developers/tanliboy.json b/data/developers/tanliboy.json index daa17945601d8528cbfc7c883e4dc4315ea363fc..7b17e651b33310ae087a6456408950e1456ff623 100644 --- a/data/developers/tanliboy.json +++ b/data/developers/tanliboy.json @@ -7,12 +7,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4501, - "hfopenllm_v2/BBH": 0.5472, - "hfopenllm_v2/MATH Level 5": 0.0944, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4017, - "hfopenllm_v2/MMLU-PRO": 0.3792 + "hfopenllm_v2/IFEval": 0.1829, + "hfopenllm_v2/BBH": 0.5488, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.4056, + "hfopenllm_v2/MMLU-PRO": 0.3805 } }, { diff --git a/data/developers/valiantlabs.json b/data/developers/valiantlabs.json index fea79922ada6f8587a14d0c683f2c0090d2ca061..a0fc2b3d81b7ddfe31ae4c63f4f370e45c708501 100644 --- a/data/developers/valiantlabs.json +++ b/data/developers/valiantlabs.json @@ -105,12 +105,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6496, - "hfopenllm_v2/BBH": 0.4774, - "hfopenllm_v2/MATH Level 5": 0.0566, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.3909, - "hfopenllm_v2/MMLU-PRO": 0.3382 + "hfopenllm_v2/IFEval": 0.2678, + "hfopenllm_v2/BBH": 0.4429, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.302, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.2927 } }, { diff --git a/data/developers/weqweasdas.json b/data/developers/weqweasdas.json index 36aa4e34297ffab30b15be8c96f1b385a32071bf..e11faa49d0109edbe3a144d975a897042bc51bca 100644 --- a/data/developers/weqweasdas.json +++ b/data/developers/weqweasdas.json @@ -7,17 +7,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2498, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Safety": 0.24, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564, + "reward-bench/Score": 0.5027, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, + "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315 + "reward-bench/Ties": 0.0315, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564 } }, { @@ -45,17 +45,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4826, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.4978, - "reward-bench/Safety": 0.4822, - "reward-bench/Reasoning": 0.7362, - "reward-bench/Prior Sets (0.5 weight)": 0.7069, + "reward-bench/Score": 0.6967, "reward-bench/Factuality": 0.4926, "reward-bench/Precise IF": 0.3937, "reward-bench/Math": 0.6066, + "reward-bench/Safety": 0.5784, "reward-bench/Focus": 0.497, - "reward-bench/Ties": 0.4232 + "reward-bench/Ties": 0.4232, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.4978, + "reward-bench/Reasoning": 0.7362, + "reward-bench/Prior Sets (0.5 weight)": 0.7069 } }, { diff --git a/data/developers/xai.json b/data/developers/xai.json index d0ba30a6d3c372a4bdb6c48d3c0cb90677ed30d5..4539ab8505af399ecb6f06dee4d4f2618de3e3bc 100644 --- a/data/developers/xai.json +++ b/data/developers/xai.json @@ -120,7 +120,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 14.2 + "terminal-bench-2.0/terminal-bench-2.0": 25.8 } } ] diff --git a/data/developers/yam-peleg.json b/data/developers/yam-peleg.json index 3161f95274f1998cddb90c53a59dce4097dbe0fd..f415128b8e93253c12ae15436b68cf36d5f248bf 100644 --- a/data/developers/yam-peleg.json +++ b/data/developers/yam-peleg.json @@ -35,12 +35,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } } ] diff --git a/data/models.json b/data/models.json index b08337869e0315ca25ad74e1e682be0126b52503..a5af0007574aa89e984bf7d8a4b2911e04bfd788 100644 --- a/data/models.json +++ b/data/models.json @@ -1005,12 +1005,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1746, - "hfopenllm_v2/BBH": 0.3126, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.245, - "hfopenllm_v2/MUSR": 0.4096, - "hfopenllm_v2/MMLU-PRO": 0.1087 + "hfopenllm_v2/IFEval": 0.1685, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.3963, + "hfopenllm_v2/MMLU-PRO": 0.1066 } }, { @@ -1391,9 +1391,9 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7008, - "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3882, + "reward-bench/Score": 0.6924, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.3575, "reward-bench/Safety": 0.7757 } }, @@ -2390,17 +2390,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9021, + "reward-bench/Score": 0.7606, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Safety": 0.8844, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8835 } }, { @@ -2447,12 +2447,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8379, - "hfopenllm_v2/BBH": 0.6157, - "hfopenllm_v2/MATH Level 5": 0.3829, + "hfopenllm_v2/IFEval": 0.8291, + "hfopenllm_v2/BBH": 0.6164, + "hfopenllm_v2/MATH Level 5": 0.4502, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4988, - "hfopenllm_v2/MMLU-PRO": 0.4656 + "hfopenllm_v2/MUSR": 0.4948, + "hfopenllm_v2/MMLU-PRO": 0.4645 } }, { @@ -2508,12 +2508,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8267, - "hfopenllm_v2/BBH": 0.405, - "hfopenllm_v2/MATH Level 5": 0.1964, - "hfopenllm_v2/GPQA": 0.2987, + "hfopenllm_v2/IFEval": 0.8255, + "hfopenllm_v2/BBH": 0.4061, + "hfopenllm_v2/MATH Level 5": 0.2115, + "hfopenllm_v2/GPQA": 0.297, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2827 + "hfopenllm_v2/MMLU-PRO": 0.2821 } }, { @@ -2555,17 +2555,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8369, + "reward-bench/Score": 0.6871, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.7588, + "reward-bench/Safety": 0.8644, + "reward-bench/Reasoning": 0.7715, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7642, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.8485, - "reward-bench/Ties": 0.6281, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.7588, - "reward-bench/Reasoning": 0.7715, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6281 } }, { @@ -6556,11 +6556,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1918, - "hfopenllm_v2/BBH": 0.2969, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2584, - "hfopenllm_v2/MUSR": 0.3846, + "hfopenllm_v2/IFEval": 0.1842, + "hfopenllm_v2/BBH": 0.2974, + "hfopenllm_v2/MATH Level 5": 0.0053, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.378, "hfopenllm_v2/MMLU-PRO": 0.1169 } }, @@ -7232,16 +7232,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.5744, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.8378, - "reward-bench/Reasoning": 0.7868, + "reward-bench/Score": 0.8008, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601 + "reward-bench/Ties": 0.5601, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Reasoning": 0.7868 } }, { @@ -7321,7 +7321,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.5 + "terminal-bench-2.0/terminal-bench-2.0": 13.9 } }, { @@ -7446,10 +7446,10 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { + "appworld_test_normal/appworld/test_normal": 0.66, "browsecompplus/browsecompplus": 0.49, - "appworld_test_normal/appworld/test_normal": 0.64, - "swe-bench/swe-bench": 0.8072, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, + "swe-bench/swe-bench": 0.65, + "tau-bench-2_airline/tau-bench-2/airline": 0.72, "tau-bench-2_retail/tau-bench-2/retail": 0.85, "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } @@ -7460,7 +7460,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 38.0 + "terminal-bench-2.0/terminal-bench-2.0": 36.9 } }, { @@ -7469,7 +7469,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 59.1 + "terminal-bench-2.0/terminal-bench-2.0": 51.9 } }, { @@ -7478,7 +7478,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 66.9 + "terminal-bench-2.0/terminal-bench-2.0": 74.7 } }, { @@ -7552,7 +7552,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 40.1 + "terminal-bench-2.0/terminal-bench-2.0": 42.6 } }, { @@ -7730,12 +7730,12 @@ "developer": "arcee-ai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5621, - "hfopenllm_v2/BBH": 0.5489, - "hfopenllm_v2/MATH Level 5": 0.2953, - "hfopenllm_v2/GPQA": 0.307, - "hfopenllm_v2/MUSR": 0.4021, - "hfopenllm_v2/MMLU-PRO": 0.3822 + "hfopenllm_v2/IFEval": 0.5718, + "hfopenllm_v2/BBH": 0.5481, + "hfopenllm_v2/MATH Level 5": 0.114, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4008, + "hfopenllm_v2/MMLU-PRO": 0.3813 } }, { @@ -8091,12 +8091,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4511, - "hfopenllm_v2/BBH": 0.4275, - "hfopenllm_v2/MATH Level 5": 0.1473, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3623, - "hfopenllm_v2/MMLU-PRO": 0.2806 + "hfopenllm_v2/IFEval": 0.4605, + "hfopenllm_v2/BBH": 0.4258, + "hfopenllm_v2/MATH Level 5": 0.0748, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3636, + "hfopenllm_v2/MMLU-PRO": 0.2812 } }, { @@ -9843,12 +9843,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.459, - "hfopenllm_v2/BBH": 0.5185, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.2743, - "hfopenllm_v2/MUSR": 0.4083, - "hfopenllm_v2/MMLU-PRO": 0.3631 + "hfopenllm_v2/IFEval": 0.7999, + "hfopenllm_v2/BBH": 0.5152, + "hfopenllm_v2/MATH Level 5": 0.1193, + "hfopenllm_v2/GPQA": 0.281, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3733 } }, { @@ -12127,12 +12127,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3613, - "hfopenllm_v2/BBH": 0.6123, - "hfopenllm_v2/MATH Level 5": 0.1239, - "hfopenllm_v2/GPQA": 0.328, - "hfopenllm_v2/MUSR": 0.4112, - "hfopenllm_v2/MMLU-PRO": 0.4494 + "hfopenllm_v2/IFEval": 0.4124, + "hfopenllm_v2/BBH": 0.6383, + "hfopenllm_v2/MATH Level 5": 0.182, + "hfopenllm_v2/GPQA": 0.3289, + "hfopenllm_v2/MUSR": 0.4349, + "hfopenllm_v2/MMLU-PRO": 0.4525 } }, { @@ -12852,12 +12852,12 @@ "developer": "Columbia-NLP", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3278, - "hfopenllm_v2/BBH": 0.392, - "hfopenllm_v2/MATH Level 5": 0.0431, - "hfopenllm_v2/GPQA": 0.2492, - "hfopenllm_v2/MUSR": 0.412, - "hfopenllm_v2/MMLU-PRO": 0.1666 + "hfopenllm_v2/IFEval": 0.3102, + "hfopenllm_v2/BBH": 0.3881, + "hfopenllm_v2/MATH Level 5": 0.0536, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.4081, + "hfopenllm_v2/MMLU-PRO": 0.1665 } }, { @@ -13400,12 +13400,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1916, - "hfopenllm_v2/BBH": 0.2977, - "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/IFEval": 0.1949, + "hfopenllm_v2/BBH": 0.2965, + "hfopenllm_v2/MATH Level 5": 0.0045, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3872, - "hfopenllm_v2/MMLU-PRO": 0.1132 + "hfopenllm_v2/MUSR": 0.3885, + "hfopenllm_v2/MMLU-PRO": 0.1111 } }, { @@ -14226,12 +14226,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.777, - "hfopenllm_v2/BBH": 0.5187, - "hfopenllm_v2/MATH Level 5": 0.2198, - "hfopenllm_v2/GPQA": 0.2936, - "hfopenllm_v2/MUSR": 0.3911, - "hfopenllm_v2/MMLU-PRO": 0.3738 + "hfopenllm_v2/IFEval": 0.5064, + "hfopenllm_v2/BBH": 0.5112, + "hfopenllm_v2/MATH Level 5": 0.1631, + "hfopenllm_v2/GPQA": 0.3163, + "hfopenllm_v2/MUSR": 0.3973, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { @@ -14338,12 +14338,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4855, - "hfopenllm_v2/BBH": 0.6627, - "hfopenllm_v2/MATH Level 5": 0.4841, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4256, - "hfopenllm_v2/MMLU-PRO": 0.5542 + "hfopenllm_v2/IFEval": 0.3745, + "hfopenllm_v2/BBH": 0.6668, + "hfopenllm_v2/MATH Level 5": 0.4758, + "hfopenllm_v2/GPQA": 0.3943, + "hfopenllm_v2/MUSR": 0.4858, + "hfopenllm_v2/MMLU-PRO": 0.5593 } }, { @@ -15729,12 +15729,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7917, - "hfopenllm_v2/BBH": 0.5109, - "hfopenllm_v2/MATH Level 5": 0.1088, - "hfopenllm_v2/GPQA": 0.2878, - "hfopenllm_v2/MUSR": 0.4136, - "hfopenllm_v2/MMLU-PRO": 0.3876 + "hfopenllm_v2/IFEval": 0.5365, + "hfopenllm_v2/BBH": 0.517, + "hfopenllm_v2/MATH Level 5": 0.1707, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4487, + "hfopenllm_v2/MMLU-PRO": 0.396 } }, { @@ -17020,12 +17020,12 @@ "developer": "DoppelReflEx", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.451, - "hfopenllm_v2/BBH": 0.4944, - "hfopenllm_v2/MATH Level 5": 0.1156, - "hfopenllm_v2/GPQA": 0.3196, - "hfopenllm_v2/MUSR": 0.3896, - "hfopenllm_v2/MMLU-PRO": 0.3256 + "hfopenllm_v2/IFEval": 0.436, + "hfopenllm_v2/BBH": 0.4956, + "hfopenllm_v2/MATH Level 5": 0.0589, + "hfopenllm_v2/GPQA": 0.3205, + "hfopenllm_v2/MUSR": 0.3843, + "hfopenllm_v2/MMLU-PRO": 0.3237 } }, { @@ -20340,12 +20340,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7207, - "hfopenllm_v2/BBH": 0.461, - "hfopenllm_v2/MATH Level 5": 0.1314, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3432, - "hfopenllm_v2/MMLU-PRO": 0.3354 + "hfopenllm_v2/IFEval": 0.7305, + "hfopenllm_v2/BBH": 0.4649, + "hfopenllm_v2/MATH Level 5": 0.1397, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3209, + "hfopenllm_v2/MMLU-PRO": 0.348 } }, { @@ -21040,12 +21040,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4672, - "hfopenllm_v2/BBH": 0.5013, - "hfopenllm_v2/MATH Level 5": 0.0279, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.386, - "hfopenllm_v2/MMLU-PRO": 0.3482 + "hfopenllm_v2/IFEval": 0.6106, + "hfopenllm_v2/BBH": 0.5347, + "hfopenllm_v2/MATH Level 5": 0.1548, + "hfopenllm_v2/GPQA": 0.3146, + "hfopenllm_v2/MUSR": 0.3991, + "hfopenllm_v2/MMLU-PRO": 0.3752 } }, { @@ -21500,12 +21500,12 @@ "developer": "fblgit", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5181, - "hfopenllm_v2/BBH": 0.7033, - "hfopenllm_v2/MATH Level 5": 0.4947, - "hfopenllm_v2/GPQA": 0.3826, - "hfopenllm_v2/MUSR": 0.5008, - "hfopenllm_v2/MMLU-PRO": 0.5915 + "hfopenllm_v2/IFEval": 0.4503, + "hfopenllm_v2/BBH": 0.7035, + "hfopenllm_v2/MATH Level 5": 0.3943, + "hfopenllm_v2/GPQA": 0.401, + "hfopenllm_v2/MUSR": 0.5021, + "hfopenllm_v2/MMLU-PRO": 0.5911 } }, { @@ -23289,12 +23289,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7598, - "hfopenllm_v2/BBH": 0.5107, - "hfopenllm_v2/MATH Level 5": 0.4237, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.4539, - "hfopenllm_v2/MMLU-PRO": 0.4012 + "hfopenllm_v2/IFEval": 0.7628, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2802, + "hfopenllm_v2/MUSR": 0.4579, + "hfopenllm_v2/MMLU-PRO": 0.4033 } }, { @@ -23519,6 +23519,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -23526,7 +23527,6 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, - "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -24103,7 +24103,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 17.1 + "terminal-bench-2.0/terminal-bench-2.0": 15.4 } }, { @@ -24203,7 +24203,7 @@ "reward-bench/Safety": 0.881, "reward-bench/Focus": 0.805, "reward-bench/Ties": 0.811, - "terminal-bench-2.0/terminal-bench-2.0": 19.6 + "terminal-bench-2.0/terminal-bench-2.0": 16.4 } }, { @@ -24259,7 +24259,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.13, + "appworld_test_normal/appworld/test_normal": 0.36, "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, @@ -24281,9 +24281,9 @@ "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.7, - "tau-bench-2_retail/tau-bench-2/retail": 0.7805, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 + "tau-bench-2_airline/tau-bench-2/airline": 0.68, + "tau-bench-2_retail/tau-bench-2/retail": 0.82, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 } }, { @@ -24292,7 +24292,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 78.4 + "terminal-bench-2.0/terminal-bench-2.0": 74.8 } }, { @@ -25558,12 +25558,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4441, - "hfopenllm_v2/BBH": 0.4863, + "hfopenllm_v2/IFEval": 0.288, + "hfopenllm_v2/BBH": 0.5154, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.307, - "hfopenllm_v2/MUSR": 0.3986, - "hfopenllm_v2/MMLU-PRO": 0.3098 + "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/MUSR": 0.408, + "hfopenllm_v2/MMLU-PRO": 0.3817 } }, { @@ -25894,17 +25894,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7847, + "reward-bench/Score": 0.5851, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Safety": 0.6956, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, - "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508 + "reward-bench/Ties": 0.5988 } }, { @@ -26814,12 +26814,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3842, - "hfopenllm_v2/BBH": 0.3144, - "hfopenllm_v2/MATH Level 5": 0.0151, - "hfopenllm_v2/GPQA": 0.255, - "hfopenllm_v2/MUSR": 0.3461, - "hfopenllm_v2/MMLU-PRO": 0.1117 + "hfopenllm_v2/IFEval": 0.083, + "hfopenllm_v2/BBH": 0.3053, + "hfopenllm_v2/MATH Level 5": 0.0083, + "hfopenllm_v2/GPQA": 0.2651, + "hfopenllm_v2/MUSR": 0.3423, + "hfopenllm_v2/MMLU-PRO": 0.1126 } }, { @@ -28507,16 +28507,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7648, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Safety": 0.9644, - "reward-bench/Reasoning": 0.9912, + "reward-bench/Score": 0.9511, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622 + "reward-bench/Ties": 0.8622, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Reasoning": 0.9912 } }, { @@ -28651,16 +28651,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -28669,16 +28669,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9016, + "reward-bench/Score": 0.5628, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Safety": 0.6111, + "reward-bench/Reasoning": 0.9576, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, - "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Reasoning": 0.9576 + "reward-bench/Ties": 0.5483 } }, { @@ -28985,12 +28985,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2477, - "hfopenllm_v2/BBH": 0.4758, - "hfopenllm_v2/MATH Level 5": 0.0453, - "hfopenllm_v2/GPQA": 0.2911, - "hfopenllm_v2/MUSR": 0.3641, - "hfopenllm_v2/MMLU-PRO": 0.3292 + "hfopenllm_v2/IFEval": 0.2553, + "hfopenllm_v2/BBH": 0.4725, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.2919, + "hfopenllm_v2/MUSR": 0.3654, + "hfopenllm_v2/MMLU-PRO": 0.3316 } }, { @@ -30623,12 +30623,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4345, - "hfopenllm_v2/BBH": 0.5419, - "hfopenllm_v2/MATH Level 5": 0.1292, - "hfopenllm_v2/GPQA": 0.3087, + "hfopenllm_v2/IFEval": 0.4418, + "hfopenllm_v2/BBH": 0.5406, + "hfopenllm_v2/MATH Level 5": 0.1352, + "hfopenllm_v2/GPQA": 0.3062, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.3854 + "hfopenllm_v2/MMLU-PRO": 0.386 } }, { @@ -35901,12 +35901,12 @@ "developer": "kavonalds", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2701, - "hfopenllm_v2/BBH": 0.5566, + "hfopenllm_v2/IFEval": 0.3283, + "hfopenllm_v2/BBH": 0.6651, "hfopenllm_v2/MATH Level 5": 0.068, - "hfopenllm_v2/GPQA": 0.2802, - "hfopenllm_v2/MUSR": 0.3682, - "hfopenllm_v2/MMLU-PRO": 0.1449 + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3393, + "hfopenllm_v2/MMLU-PRO": 0.1314 } }, { @@ -39569,16 +39569,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9294, + "reward-bench/Score": 0.7394, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Safety": 0.9178, + "reward-bench/Reasoning": 0.9698, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Reasoning": 0.9698 + "reward-bench/Ties": 0.7653 } }, { @@ -39741,12 +39741,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4118, - "hfopenllm_v2/BBH": 0.4811, - "hfopenllm_v2/MATH Level 5": 0.034, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3047, - "hfopenllm_v2/MMLU-PRO": 0.3006 + "hfopenllm_v2/IFEval": 0.4027, + "hfopenllm_v2/BBH": 0.4789, + "hfopenllm_v2/MATH Level 5": 0.0461, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.3087, + "hfopenllm_v2/MMLU-PRO": 0.3001 } }, { @@ -42056,12 +42056,12 @@ "developer": "meta-llama", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4782, - "hfopenllm_v2/BBH": 0.491, - "hfopenllm_v2/MATH Level 5": 0.0914, - "hfopenllm_v2/GPQA": 0.2928, - "hfopenllm_v2/MUSR": 0.3805, - "hfopenllm_v2/MMLU-PRO": 0.3591, + "hfopenllm_v2/IFEval": 0.7408, + "hfopenllm_v2/BBH": 0.4989, + "hfopenllm_v2/MATH Level 5": 0.0869, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3568, + "hfopenllm_v2/MMLU-PRO": 0.3664, "reward-bench/Score": 0.645, "reward-bench/Chat": 0.8547, "reward-bench/Chat Hard": 0.4156, @@ -44247,12 +44247,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6283, - "hfopenllm_v2/BBH": 0.583, - "hfopenllm_v2/MATH Level 5": 0.2039, - "hfopenllm_v2/GPQA": 0.3331, - "hfopenllm_v2/MUSR": 0.4063, - "hfopenllm_v2/MMLU-PRO": 0.4099 + "hfopenllm_v2/IFEval": 0.667, + "hfopenllm_v2/BBH": 0.5213, + "hfopenllm_v2/MATH Level 5": 0.1435, + "hfopenllm_v2/GPQA": 0.3238, + "hfopenllm_v2/MUSR": 0.3632, + "hfopenllm_v2/MMLU-PRO": 0.396 } }, { @@ -44758,12 +44758,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7561, - "hfopenllm_v2/BBH": 0.5111, - "hfopenllm_v2/MATH Level 5": 0.0906, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3841 + "hfopenllm_v2/IFEval": 0.4162, + "hfopenllm_v2/BBH": 0.5124, + "hfopenllm_v2/MATH Level 5": 0.0853, + "hfopenllm_v2/GPQA": 0.3029, + "hfopenllm_v2/MUSR": 0.415, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { @@ -44996,7 +44996,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 26.7 + "terminal-bench-2.0/terminal-bench-2.0": 27.8 } }, { @@ -45317,7 +45317,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 58.4 + "terminal-bench-2.0/terminal-bench-2.0": 72.4 } }, { @@ -46416,16 +46416,16 @@ "developer": "NCSOFT", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8942, + "reward-bench/Score": 0.648, + "reward-bench/Chat": 0.9721, + "reward-bench/Chat Hard": 0.818, + "reward-bench/Safety": 0.7222, + "reward-bench/Reasoning": 0.9192, "reward-bench/Factuality": 0.6084, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.9596, - "reward-bench/Ties": 0.6786, - "reward-bench/Chat": 0.9721, - "reward-bench/Chat Hard": 0.818, - "reward-bench/Reasoning": 0.9192 + "reward-bench/Ties": 0.6786 } }, { @@ -48142,17 +48142,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4553, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Safety": 0.7556, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137, + "reward-bench/Score": 0.8133, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, + "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004 + "reward-bench/Ties": 0.1004, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137 } }, { @@ -50393,12 +50393,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { @@ -51011,16 +51011,16 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { + "ace/Overall Score": 0.515, + "ace/Food Score": 0.65, + "ace/Gaming Score": 0.578, "apex-agents/Overall Pass@1": 0.23, "apex-agents/Overall Pass@8": 0.4, "apex-agents/Overall Mean Score": 0.387, "apex-agents/Investment Banking Pass@1": 0.273, "apex-agents/Management Consulting Pass@1": 0.227, "apex-agents/Corporate Law Pass@1": 0.189, - "apex-agents/Corporate Lawyer Mean Score": 0.443, - "ace/Overall Score": 0.515, - "ace/Food Score": 0.65, - "ace/Gaming Score": 0.578 + "apex-agents/Corporate Lawyer Mean Score": 0.443 } }, { @@ -51620,16 +51620,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.8673, + "reward-bench/Score": 0.6493, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.8619, + "reward-bench/Reasoning": 0.8661, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, - "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.8661 + "reward-bench/Ties": 0.7819 } }, { @@ -51707,16 +51707,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 } }, { @@ -51770,7 +51770,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 41.3 + "terminal-bench-2.0/terminal-bench-2.0": 43.4 } }, { @@ -51779,7 +51779,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 22.2 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -51802,7 +51802,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.9 + "terminal-bench-2.0/terminal-bench-2.0": 7.0 } }, { @@ -51834,7 +51834,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 + "terminal-bench-2.0/terminal-bench-2.0": 53.5 } }, { @@ -51861,7 +51861,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 54.0 + "terminal-bench-2.0/terminal-bench-2.0": 60.7 } }, { @@ -51870,15 +51870,15 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.48, + "appworld_test_normal/appworld/test_normal": 0.071, + "browsecompplus/browsecompplus": 0.46, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.5455, - "tau-bench-2_airline/tau-bench-2/airline": 0.48, - "tau-bench-2_retail/tau-bench-2/retail": 0.51, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 + "swe-bench/swe-bench": 0.57, + "tau-bench-2_airline/tau-bench-2/airline": 0.6, + "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 } }, { @@ -51896,7 +51896,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 77.3 + "terminal-bench-2.0/terminal-bench-2.0": 74.6 } }, { @@ -51978,7 +51978,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.056338028169014086, "livecodebenchpro/Easy Problems": 0.5070422535211268, - "terminal-bench-2.0/terminal-bench-2.0": 3.4 + "terminal-bench-2.0/terminal-bench-2.0": 3.1 } }, { @@ -52279,17 +52279,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.32, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Safety": 0.3667, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836, + "reward-bench/Score": 0.6126, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, + "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12 + "reward-bench/Ties": 0.12, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836 } }, { @@ -52312,17 +52312,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5806, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Safety": 0.6267, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172, + "reward-bench/Score": 0.8159, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972 + "reward-bench/Ties": 0.5972, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172 } }, { @@ -53830,17 +53830,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3332, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Safety": 0.7589, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57, + "reward-bench/Score": 0.5798, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, + "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57 } }, { @@ -54983,12 +54983,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6052, - "hfopenllm_v2/BBH": 0.6317, - "hfopenllm_v2/MATH Level 5": 0.4789, - "hfopenllm_v2/GPQA": 0.3742, - "hfopenllm_v2/MUSR": 0.486, - "hfopenllm_v2/MMLU-PRO": 0.5302 + "hfopenllm_v2/IFEval": 0.6064, + "hfopenllm_v2/BBH": 0.6296, + "hfopenllm_v2/MATH Level 5": 0.3708, + "hfopenllm_v2/GPQA": 0.3733, + "hfopenllm_v2/MUSR": 0.4873, + "hfopenllm_v2/MMLU-PRO": 0.5307 } }, { @@ -56605,12 +56605,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { @@ -59400,17 +59400,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8464, + "reward-bench/Score": 0.589, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Safety": 0.7222, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209 + "reward-bench/Ties": 0.5743 } }, { @@ -59419,16 +59419,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9154, + "reward-bench/Score": 0.6766, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8618, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9362, "reward-bench/Factuality": 0.6274, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.5847, - "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.8929, - "reward-bench/Ties": 0.6824, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8618, - "reward-bench/Reasoning": 0.9362 + "reward-bench/Ties": 0.6824 } }, { @@ -59437,17 +59437,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8542, + "reward-bench/Score": 0.6089, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Safety": 0.7867, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309 + "reward-bench/Ties": 0.5981 } }, { @@ -59511,12 +59511,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2854, - "hfopenllm_v2/BBH": 0.5984, - "hfopenllm_v2/MATH Level 5": 0.1005, - "hfopenllm_v2/GPQA": 0.3297, - "hfopenllm_v2/MUSR": 0.4607, - "hfopenllm_v2/MMLU-PRO": 0.4162 + "hfopenllm_v2/IFEval": 0.7649, + "hfopenllm_v2/BBH": 0.5974, + "hfopenllm_v2/MATH Level 5": 0.0174, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4245, + "hfopenllm_v2/MMLU-PRO": 0.4207 } }, { @@ -59539,12 +59539,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2747, - "hfopenllm_v2/BBH": 0.6031, - "hfopenllm_v2/MATH Level 5": 0.0831, - "hfopenllm_v2/GPQA": 0.3305, - "hfopenllm_v2/MUSR": 0.4686, - "hfopenllm_v2/MMLU-PRO": 0.4122 + "hfopenllm_v2/IFEval": 0.7592, + "hfopenllm_v2/BBH": 0.6026, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.3289, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.4163 } }, { @@ -59553,12 +59553,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7439, - "hfopenllm_v2/BBH": 0.5993, - "hfopenllm_v2/MATH Level 5": 0.0876, - "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4204, - "hfopenllm_v2/MMLU-PRO": 0.4072 + "hfopenllm_v2/IFEval": 0.5761, + "hfopenllm_v2/BBH": 0.602, + "hfopenllm_v2/MATH Level 5": 0.1888, + "hfopenllm_v2/GPQA": 0.3372, + "hfopenllm_v2/MUSR": 0.4632, + "hfopenllm_v2/MMLU-PRO": 0.4039 } }, { @@ -59721,12 +59721,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0932, - "hfopenllm_v2/BBH": 0.2977, + "hfopenllm_v2/IFEval": 0.0905, + "hfopenllm_v2/BBH": 0.2985, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2475, - "hfopenllm_v2/MUSR": 0.3941, - "hfopenllm_v2/MMLU-PRO": 0.1157 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3848, + "hfopenllm_v2/MMLU-PRO": 0.1158 } }, { @@ -59861,12 +59861,12 @@ "developer": "riaz", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4373, - "hfopenllm_v2/BBH": 0.4586, - "hfopenllm_v2/MATH Level 5": 0.0514, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3763, - "hfopenllm_v2/MMLU-PRO": 0.2964 + "hfopenllm_v2/IFEval": 0.4137, + "hfopenllm_v2/BBH": 0.4565, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.276, + "hfopenllm_v2/MUSR": 0.3776, + "hfopenllm_v2/MMLU-PRO": 0.2978 } }, { @@ -60158,12 +60158,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2566, - "hfopenllm_v2/BBH": 0.39, - "hfopenllm_v2/MATH Level 5": 0.1208, - "hfopenllm_v2/GPQA": 0.2626, + "hfopenllm_v2/IFEval": 0.2595, + "hfopenllm_v2/BBH": 0.3884, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2743, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2741 + "hfopenllm_v2/MMLU-PRO": 0.2719 } }, { @@ -61835,12 +61835,12 @@ "developer": "Sao10K", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7281, - "hfopenllm_v2/BBH": 0.6503, - "hfopenllm_v2/MATH Level 5": 0.2243, + "hfopenllm_v2/IFEval": 0.7384, + "hfopenllm_v2/BBH": 0.6471, + "hfopenllm_v2/MATH Level 5": 0.2137, "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4196, - "hfopenllm_v2/MMLU-PRO": 0.5096 + "hfopenllm_v2/MUSR": 0.4209, + "hfopenllm_v2/MMLU-PRO": 0.5104 } }, { @@ -62581,16 +62581,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } }, { @@ -63307,16 +63307,16 @@ "hfopenllm_v2/GPQA": 0.344, "hfopenllm_v2/MUSR": 0.4231, "hfopenllm_v2/MMLU-PRO": 0.4103, - "reward-bench/Score": 0.7531, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.8991, - "reward-bench/Safety": 0.9689, - "reward-bench/Reasoning": 0.9807, + "reward-bench/Score": 0.9426, "reward-bench/Factuality": 0.7674, "reward-bench/Precise IF": 0.375, "reward-bench/Math": 0.6721, + "reward-bench/Safety": 0.9297, "reward-bench/Focus": 0.9172, - "reward-bench/Ties": 0.8182 + "reward-bench/Ties": 0.8182, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.8991, + "reward-bench/Reasoning": 0.9807 } }, { @@ -66784,12 +66784,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4501, - "hfopenllm_v2/BBH": 0.5472, - "hfopenllm_v2/MATH Level 5": 0.0944, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4017, - "hfopenllm_v2/MMLU-PRO": 0.3792 + "hfopenllm_v2/IFEval": 0.1829, + "hfopenllm_v2/BBH": 0.5488, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.4056, + "hfopenllm_v2/MMLU-PRO": 0.3805 } }, { @@ -71056,12 +71056,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6496, - "hfopenllm_v2/BBH": 0.4774, - "hfopenllm_v2/MATH Level 5": 0.0566, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.3909, - "hfopenllm_v2/MMLU-PRO": 0.3382 + "hfopenllm_v2/IFEval": 0.2678, + "hfopenllm_v2/BBH": 0.4429, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.302, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.2927 } }, { @@ -71686,17 +71686,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2498, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Safety": 0.24, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564, + "reward-bench/Score": 0.5027, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, + "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315 + "reward-bench/Ties": 0.0315, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564 } }, { @@ -71724,17 +71724,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4826, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.4978, - "reward-bench/Safety": 0.4822, - "reward-bench/Reasoning": 0.7362, - "reward-bench/Prior Sets (0.5 weight)": 0.7069, + "reward-bench/Score": 0.6967, "reward-bench/Factuality": 0.4926, "reward-bench/Precise IF": 0.3937, "reward-bench/Math": 0.6066, + "reward-bench/Safety": 0.5784, "reward-bench/Focus": 0.497, - "reward-bench/Ties": 0.4232 + "reward-bench/Ties": 0.4232, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.4978, + "reward-bench/Reasoning": 0.7362, + "reward-bench/Prior Sets (0.5 weight)": 0.7069 } }, { @@ -72436,7 +72436,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 14.2 + "terminal-bench-2.0/terminal-bench-2.0": 25.8 } }, { @@ -73047,12 +73047,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } }, { diff --git a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json index 776a7c662b5afb276bf059bb90896cee5cd968ec..a8a69697e53329e56da6b7741229b84b4f2eeb85 100644 --- a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json +++ b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json @@ -5,7 +5,7 @@ "developer": "adriszmar", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1685 + "score": 0.1746 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3124 + "score": 0.3126 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0015 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2492 + "score": 0.245 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3963 + "score": 0.4096 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1066 + "score": 0.1087 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1746 + "score": 0.1685 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3126 + "score": 0.3124 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0015 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.245 + "score": 0.2492 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4096 + "score": 0.3963 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1087 + "score": 0.1066 } } ], diff --git a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json index e1ac41e4bb97f41e9784803c588108ddfee89905..e9d556b28d7550c1e76a8e996c98b852c7622823 100644 --- a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json +++ b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json @@ -38,7 +38,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7058 + "score": 0.6905 }, "source_data": { "dataset_name": "RewardBench", @@ -56,7 +56,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9525 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -74,7 +74,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3947 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -92,7 +92,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7703 + "score": 0.7676 }, "source_data": { "dataset_name": "RewardBench", @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6895 + "score": 0.6808 }, "source_data": { "dataset_name": "RewardBench", @@ -152,7 +152,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9302 }, "source_data": { "dataset_name": "RewardBench", @@ -170,7 +170,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -188,7 +188,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7595 + "score": 0.7527 }, "source_data": { "dataset_name": "RewardBench", @@ -230,7 +230,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7019 + "score": 0.6895 }, "source_data": { "dataset_name": "RewardBench", @@ -248,7 +248,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -266,7 +266,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -284,7 +284,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7811 + "score": 0.7595 }, "source_data": { "dataset_name": "RewardBench", @@ -326,7 +326,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6905 + "score": 0.7004 }, "source_data": { "dataset_name": "RewardBench", @@ -344,7 +344,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9413 }, "source_data": { "dataset_name": "RewardBench", @@ -362,7 +362,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -380,7 +380,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7676 + "score": 0.7716 }, "source_data": { "dataset_name": "RewardBench", @@ -422,7 +422,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6808 + "score": 0.7019 }, "source_data": { "dataset_name": "RewardBench", @@ -440,7 +440,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.9497 }, "source_data": { "dataset_name": "RewardBench", @@ -458,7 +458,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.375 }, "source_data": { "dataset_name": "RewardBench", @@ -476,7 +476,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7527 + "score": 0.7811 }, "source_data": { "dataset_name": "RewardBench", @@ -518,7 +518,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6924 + "score": 0.7058 }, "source_data": { "dataset_name": "RewardBench", @@ -536,7 +536,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9525 }, "source_data": { "dataset_name": "RewardBench", @@ -554,7 +554,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3575 + "score": 0.3947 }, "source_data": { "dataset_name": "RewardBench", @@ -572,7 +572,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7703 }, "source_data": { "dataset_name": "RewardBench", @@ -614,7 +614,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6945 + "score": 0.7008 }, "source_data": { "dataset_name": "RewardBench", @@ -650,7 +650,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -668,7 +668,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7743 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -710,7 +710,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7004 + "score": 0.6945 }, "source_data": { "dataset_name": "RewardBench", @@ -728,7 +728,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9413 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -746,7 +746,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -764,7 +764,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7716 + "score": 0.7743 }, "source_data": { "dataset_name": "RewardBench", @@ -806,7 +806,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7008 + "score": 0.6924 }, "source_data": { "dataset_name": "RewardBench", @@ -824,7 +824,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -842,7 +842,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3575 }, "source_data": { "dataset_name": "RewardBench", diff --git a/data/models/alibaba_qwen-3-coder-480b.json b/data/models/alibaba_qwen-3-coder-480b.json index f598d15a400b5ed438e43880c6cab3a3248e3641..78f6ba994c4cff4ccaf572ddc5297bedb19efb9b 100644 --- a/data/models/alibaba_qwen-3-coder-480b.json +++ b/data/models/alibaba_qwen-3-coder-480b.json @@ -4,13 +4,13 @@ "id": "alibaba/qwen-3-coder-480b", "developer": "Alibaba", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Dakou Agent", + "agent_organization": "iflow" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-12-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 25.4, + "score": 27.2, "uncertainty": { "standard_error": { "value": 2.6 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-28", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 25.4, "uncertainty": { "standard_error": { "value": 2.6 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json b/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json index 6b82bed55df1bdb62067f41e1f2f588c7fd5e772..3edb4e861c9c885f1d8b857f90568f09968d9fbc 100644 --- a/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json +++ b/data/models/allenai_llama-3.1-70b-instruct-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.7606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8126 + "score": 0.9021 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.8355 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8844 + "score": 0.9095 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8646 + "score": 0.8969 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8835 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9021 + "score": 0.7606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.8126 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8355 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6995 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9095 + "score": 0.8844 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8969 + "score": 0.8646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8835 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-70b.json b/data/models/allenai_llama-3.1-tulu-3-70b.json index 1db7ede127ee803b7ffd5960ed83ddaa8fe5c58c..dedd220cd13dda3e92225bd87465795dab977b4c 100644 --- a/data/models/allenai_llama-3.1-tulu-3-70b.json +++ b/data/models/allenai_llama-3.1-tulu-3-70b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "70.554" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8291 + "score": 0.8379 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6164 + "score": 0.6157 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4502 + "score": 0.3829 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4948 + "score": 0.4988 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4645 + "score": 0.4656 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8379 + "score": 0.8291 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6157 + "score": 0.6164 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3829 + "score": 0.4502 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4988 + "score": 0.4948 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4656 + "score": 0.4645 } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b-rl-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-8b-rl-rm-rb2.json index c04edea77c13058100e53f5dc2eead6fb95e8fc2..407df5e17676737402cd08d14ff8185dc4fb9a2a 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b-rl-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b-rl-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-RL-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-RL-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6871 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7642 + "score": 0.8369 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4 + "score": 0.9469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6175 + "score": 0.7588 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8644 + "score": 0.8703 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8485 + "score": 0.7715 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6281 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-RL-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-RL-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8369 + "score": 0.6871 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9469 + "score": 0.7642 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7588 + "score": 0.4 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6175 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8703 + "score": 0.8644 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7715 + "score": 0.8485 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.6281 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b.json b/data/models/allenai_llama-3.1-tulu-3-8b.json index 53350f20b1f74556a0af6a9aa87ed77d181deaec..7de1c9431728784c04f1c32781631febc2ed7e32 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8255 + "score": 0.8267 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4061 + "score": 0.405 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2115 + "score": 0.1964 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.297 + "score": 0.2987 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2821 + "score": 0.2827 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8255 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.405 + "score": 0.4061 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1964 + "score": 0.2115 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2987 + "score": 0.297 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2821 } } ], diff --git a/data/models/amd_amd-llama-135m.json b/data/models/amd_amd-llama-135m.json index a445333e3dd0495a4c3e2fddf3e92a96b288f85c..dfdcf8e7a431ddc8bdf51b6aa797b81cf0f89c9e 100644 --- a/data/models/amd_amd-llama-135m.json +++ b/data/models/amd_amd-llama-135m.json @@ -5,9 +5,9 @@ "developer": "amd", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "0.135" + "params_billions": "0.134" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1842 + "score": 0.1918 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2974 + "score": 0.2969 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0053 + "score": 0.0076 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2525 + "score": 0.2584 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.378 + "score": 0.3846 } }, { @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1918 + "score": 0.1842 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2969 + "score": 0.2974 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0053 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2584 + "score": 0.2525 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3846 + "score": 0.378 } }, { diff --git a/data/models/anthropic_claude-3-5-haiku-20241022.json b/data/models/anthropic_claude-3-5-haiku-20241022.json index 97ab32cadbb5e110dfa409cbea1ee9b8c38d6850..9edb0c7484cd3b1eab61fa90716227281f955253 100644 --- a/data/models/anthropic_claude-3-5-haiku-20241022.json +++ b/data/models/anthropic_claude-3-5-haiku-20241022.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-opus-20240229.json b/data/models/anthropic_claude-3-opus-20240229.json index 8f8b70dd804fcbb225ceb8111209d93e20e8bd0b..150503f1a5f91571d10a33f8965d4bd84f3ccf46 100644 --- a/data/models/anthropic_claude-3-opus-20240229.json +++ b/data/models/anthropic_claude-3-opus-20240229.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,128 +1925,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8008 + "score": 0.5744 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9469 + "score": 0.5389 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.5137 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7868 + "score": 0.8378 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5744 + "score": 0.6646 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2055,111 +2031,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5389 + "score": 0.5601 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.8008 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5137 + "score": 0.9469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8378 + "score": 0.6031 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6646 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5601 + "score": 0.7868 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/anthropic_claude-haiku-4.5.json b/data/models/anthropic_claude-haiku-4.5.json index 0030689870b1d61c70e99ee0cdff0afc1d0e9bf2..e7b7ed8c329f7f81d319b4442fafacedb7ba1133 100644 --- a/data/models/anthropic_claude-haiku-4.5.json +++ b/data/models/anthropic_claude-haiku-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-haiku-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 28.3, + "score": 29.8, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.5, + "score": 28.3, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.8, + "score": 35.5, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 13.9, + "score": 27.5, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.5, + "score": 13.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4-5.json b/data/models/anthropic_claude-opus-4-5.json index 0cd69b8f1e682224797162bd821dc65c4792b021..ec98797a0bada43df823aeb8285704dc256b7f56 100644 --- a/data/models/anthropic_claude-opus-4-5.json +++ b/data/models/anthropic_claude-opus-4-5.json @@ -10,7 +10,7 @@ }, "evaluations": [ { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -23,34 +23,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.64, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "3.43", + "total_run_cost": "343.32", + "average_steps": "20.06", + "percent_finished": "0.82" } }, "generation_config": { @@ -78,7 +78,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.7, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "11.32", - "total_run_cost": "1132.47", - "average_steps": "21.99", - "percent_finished": "0.83" + "average_agent_cost": "5.59", + "total_run_cost": "558.51", + "average_steps": "41.07", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -159,42 +159,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5294, + "score": 0.61, "uncertainty": { - "num_samples": 51 + "num_samples": 100 }, "details": { - "average_agent_cost": "11.66", - "total_run_cost": "594.68", - "average_steps": "31.04", - "percent_finished": "0.8431" + "average_agent_cost": "11.32", + "total_run_cost": "1132.47", + "average_steps": "21.99", + "percent_finished": "0.83" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -206,8 +206,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -282,7 +282,7 @@ } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -295,34 +295,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.5294, "uncertainty": { - "num_samples": 100 + "num_samples": 51 }, "details": { - "average_agent_cost": "13.08", - "total_run_cost": "1308.38", - "average_steps": "49.69", - "percent_finished": "0.74" + "average_agent_cost": "11.66", + "total_run_cost": "594.68", + "average_steps": "31.04", + "percent_finished": "0.8431" } }, "generation_config": { @@ -349,6 +349,74 @@ } } }, + { + "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "retrieved_timestamp": "1774263615.0201504", + "source_metadata": { + "source_name": "Exgentic Open Agent Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Exgentic", + "source_organization_url": "https://github.com/Exgentic", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "exgentic", + "version": "0.1.0" + }, + "benchmark": "browsecompplus", + "evaluation_results": [ + { + "evaluation_name": "browsecompplus", + "source_data": { + "dataset_name": "browsecompplus", + "source_type": "url", + "url": [ + "https://github.com/Exgentic/exgentic" + ] + }, + "metric_config": { + "evaluation_description": "BrowseCompPlus benchmark evaluation", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.49, + "uncertainty": { + "num_samples": 100 + }, + "details": { + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" + } + }, + "generation_config": { + "generation_args": { + "agentic_eval_config": { + "additional_details": { + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" + } + } + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "generation_args": { + "agentic_eval_config": { + "additional_details": { + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" + } + } + } + } + }, { "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", @@ -486,7 +554,7 @@ } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.66, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "5.59", - "total_run_cost": "558.51", - "average_steps": "41.07", - "percent_finished": "0.82" + "average_agent_cost": "13.08", + "total_run_cost": "1308.38", + "average_steps": "49.69", + "percent_finished": "0.74" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -546,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -601,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -614,15 +682,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -635,42 +703,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.64, + "score": 0.8072, "uncertainty": { - "num_samples": 100 + "num_samples": 83 }, "details": { - "average_agent_cost": "3.43", - "total_run_cost": "343.32", - "average_steps": "20.06", - "percent_finished": "0.82" + "average_agent_cost": "2.96", + "total_run_cost": "245.78", + "average_steps": "34.1", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -682,15 +750,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -737,8 +805,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -750,15 +818,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -790,14 +858,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.65, + "score": 0.7423, "uncertainty": { - "num_samples": 100 + "num_samples": 97 }, "details": { - "average_agent_cost": "4.85", - "total_run_cost": "485.22", - "average_steps": "39.13", + "average_agent_cost": "5.6", + "total_run_cost": "543.62", + "average_steps": "31.76", "percent_finished": "1.0" } }, @@ -805,8 +873,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -818,15 +886,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -873,8 +941,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -886,15 +954,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -926,14 +994,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7423, + "score": 0.65, "uncertainty": { - "num_samples": 97 + "num_samples": 100 }, "details": { - "average_agent_cost": "5.6", - "total_run_cost": "543.62", - "average_steps": "31.76", + "average_agent_cost": "4.85", + "total_run_cost": "485.22", + "average_steps": "39.13", "percent_finished": "1.0" } }, @@ -941,8 +1009,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -954,15 +1022,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -994,14 +1062,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.74, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.72", - "total_run_cost": "36.55", - "average_steps": "12.22", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1009,8 +1077,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1022,8 +1090,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1098,7 +1166,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1145,8 +1213,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1158,15 +1226,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1198,82 +1266,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.72, + "score": 0.74, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.78", - "total_run_cost": "39.67", - "average_steps": "11.88", - "percent_finished": "1.0" - } - }, - "generation_config": { - "generation_args": { - "agentic_eval_config": { - "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" - } - } - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "generation_args": { - "agentic_eval_config": { - "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" - } - } - } - } - }, - { - "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", - "retrieved_timestamp": "1774263615.0201504", - "source_metadata": { - "source_name": "Exgentic Open Agent Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Exgentic", - "source_organization_url": "https://github.com/Exgentic", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "exgentic", - "version": "0.1.0" - }, - "benchmark": "swe-bench", - "evaluation_results": [ - { - "evaluation_name": "swe-bench", - "source_data": { - "dataset_name": "swe-bench", - "source_type": "url", - "url": [ - "https://github.com/Exgentic/exgentic" - ] - }, - "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.8072, - "uncertainty": { - "num_samples": 83 - }, - "details": { - "average_agent_cost": "2.96", - "total_run_cost": "245.78", - "average_steps": "34.1", + "average_agent_cost": "0.72", + "total_run_cost": "36.55", + "average_steps": "12.22", "percent_finished": "1.0" } }, @@ -1302,7 +1302,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1334,14 +1334,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.72, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "0.78", + "total_run_cost": "39.67", + "average_steps": "11.88", "percent_finished": "1.0" } }, @@ -1349,8 +1349,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1362,15 +1362,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1407,9 +1407,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "0.67", - "total_run_cost": "68.24", - "average_steps": "11.71", + "average_agent_cost": "0.47", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1417,8 +1417,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1430,15 +1430,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1470,14 +1470,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.83, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "1.6", + "total_run_cost": "161.14", + "average_steps": "12.54", "percent_finished": "1.0" } }, @@ -1485,8 +1485,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1498,15 +1498,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1538,14 +1538,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.83, + "score": 0.78, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.6", - "total_run_cost": "161.14", - "average_steps": "12.54", + "average_agent_cost": "0.67", + "total_run_cost": "68.24", + "average_steps": "11.71", "percent_finished": "1.0" } }, @@ -1553,8 +1553,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1566,8 +1566,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1710,7 +1710,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1742,14 +1742,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "1.06", + "total_run_cost": "114.62", + "average_steps": "13.77", "percent_finished": "1.0" } }, @@ -1757,8 +1757,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1770,15 +1770,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1810,14 +1810,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.06", - "total_run_cost": "114.62", - "average_steps": "13.77", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1825,8 +1825,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1838,8 +1838,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1914,7 +1914,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1961,8 +1961,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1974,8 +1974,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } diff --git a/data/models/anthropic_claude-opus-4.1.json b/data/models/anthropic_claude-opus-4.1.json index 66d9a9691d54a3cae59a5a4b0972ad6110b06a9b..643c986c59e6df513c07edafb084900951fa2ab2 100644 --- a/data/models/anthropic_claude-opus-4.1.json +++ b/data/models/anthropic_claude-opus-4.1.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.1", "developer": "Anthropic", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Claude Code", + "agent_organization": "Anthropic" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.1, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 35.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 38.0, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 38.0, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.5.json b/data/models/anthropic_claude-opus-4.5.json index 08f15dd6e988a559ca38856b6df4c4e4d53b00a1..04c518e21852763638180858614bb6ac2be608de 100644 --- a/data/models/anthropic_claude-opus-4.5.json +++ b/data/models/anthropic_claude-opus-4.5.json @@ -78,7 +78,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -102,7 +102,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-11-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -111,11 +111,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4 + "score": 57.8, + "uncertainty": { + "standard_error": { + "value": 2.5 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -132,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -146,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -170,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-22", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -179,17 +185,11 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 58.4 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -206,7 +206,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -220,7 +220,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -244,7 +244,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -253,17 +253,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.9, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -280,7 +280,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -516,7 +516,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -540,7 +540,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -549,17 +549,17 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 51.9, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -576,7 +576,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.6.json b/data/models/anthropic_claude-opus-4.6.json index 1ae6eba958aa9144fbbc47e84b576827e9ba8012..a41a27dbfdbf4960b47a33e1930b531d6b54f85b 100644 --- a/data/models/anthropic_claude-opus-4.6.json +++ b/data/models/anthropic_claude-opus-4.6.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.6", "developer": "Anthropic", "additional_details": { - "agent_name": "TongAgents", - "agent_organization": "Bigai" + "agent_name": "Droid", + "agent_organization": "Factory" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.9, + "score": 69.9, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,11 @@ "max_score": 100.0 }, "score_details": { - "score": 69.9, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 66.9 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-13", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +185,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.7, + "score": 66.5, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +212,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +226,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +250,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +259,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, + "score": 58.0, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +300,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +324,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-13", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +333,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.5, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +398,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-07", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.0, + "score": 71.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -454,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -478,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -487,11 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.9 + "score": 74.7, + "uncertainty": { + "standard_error": { + "value": 2.6 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-sonnet-4.5.json b/data/models/anthropic_claude-sonnet-4.5.json index c0439a05bb694a022b80d498957deb384990a100..53dd8e2195a24f3d530bf69beec973cc6e94f1f4 100644 --- a/data/models/anthropic_claude-sonnet-4.5.json +++ b/data/models/anthropic_claude-sonnet-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-sonnet-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Goose", - "agent_organization": "Block" + "agent_name": "CAMEL-AI", + "agent_organization": "CAMEL-AI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.1, + "score": 46.5, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 46.5, + "score": 42.8, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.5, + "score": 40.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 42.8, + "score": 42.5, "uncertainty": { "standard_error": { "value": 2.8 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,11 @@ "max_score": 100.0 }, "score_details": { - "score": 42.6, - "uncertainty": { - "standard_error": { - "value": 2.8 - }, - "num_samples": 435 - } + "score": 42.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +398,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,11 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.7 + "score": 43.1, + "uncertainty": { + "standard_error": { + "value": 2.6 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 40.1, + "score": 42.6, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/arcee-ai_arcee-spark.json b/data/models/arcee-ai_arcee-spark.json index e194ce4efc974318b0bb01e7560254bdd047ab55..611dbc409a44fd2eadd33b4e30a3048a3002308e 100644 --- a/data/models/arcee-ai_arcee-spark.json +++ b/data/models/arcee-ai_arcee-spark.json @@ -5,7 +5,7 @@ "developer": "arcee-ai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5718 + "score": 0.5621 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5481 + "score": 0.5489 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.114 + "score": 0.2953 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.307 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4008 + "score": 0.4021 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3813 + "score": 0.3822 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5621 + "score": 0.5718 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5489 + "score": 0.5481 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2953 + "score": 0.114 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3062 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4021 + "score": 0.4008 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3822 + "score": 0.3813 } } ], diff --git a/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json b/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json index c02cc4e6043653215ef5039a1e40518d83ff39e0..d9c4a63a2048d58930bc10ecb5695a3676976b8f 100644 --- a/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json +++ b/data/models/atanddev_qwen2.5-1.5b-continuous-learnt.json @@ -5,7 +5,7 @@ "developer": "AtAndDev", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "1.544" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4605 + "score": 0.4511 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4258 + "score": 0.4275 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0748 + "score": 0.1473 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3636 + "score": 0.3623 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.2806 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4511 + "score": 0.4605 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4275 + "score": 0.4258 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1473 + "score": 0.0748 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3623 + "score": 0.3636 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2806 + "score": 0.2812 } } ], diff --git a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json index eb03ad8b35e71a4f26603a2960132d29f0246b60..ed5c0ad9e23e3e1024094c595c2931cbc579e102 100644 --- a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json +++ b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json @@ -5,7 +5,7 @@ "developer": "BoltMonkey", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7999 + "score": 0.459 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5152 + "score": 0.5185 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1193 + "score": 0.0937 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.281 + "score": 0.2743 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.4083 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3631 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.459 + "score": 0.7999 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5185 + "score": 0.5152 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.1193 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.281 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4083 + "score": 0.4019 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3631 + "score": 0.3733 } } ], diff --git a/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json b/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json index 62d09c3db567da9932caf7141d7a1389c1aafea9..9b7cc35f188c98dba3f0427de09e77d4b26cfc40 100644 --- a/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json +++ b/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json @@ -5,7 +5,7 @@ "developer": "cognitivecomputations", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "13.96" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4124 + "score": 0.3613 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6383 + "score": 0.6123 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.182 + "score": 0.1239 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.328 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4349 + "score": 0.4112 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4525 + "score": 0.4494 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3613 + "score": 0.4124 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6123 + "score": 0.6383 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1239 + "score": 0.182 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.328 + "score": 0.3289 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4112 + "score": 0.4349 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4494 + "score": 0.4525 } } ], diff --git a/data/models/cohere_command-a-03-2025.json b/data/models/cohere_command-a-03-2025.json index 205f7c496ed81b612dd2b182b3cf2e9fd9ac9c54..aedaf4e5f944a400d87afc543fa422bde24be135 100644 --- a/data/models/cohere_command-a-03-2025.json +++ b/data/models/cohere_command-a-03-2025.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json b/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json index fcba7390760a9757eb868fa82619eef91d551f86..e1500a121f56ad89301900472e223d7937764cb0 100644 --- a/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json +++ b/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json @@ -5,7 +5,7 @@ "developer": "Columbia-NLP", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "GemmaForCausalLM", "params_billions": "2.506" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3102 + "score": 0.3278 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3881 + "score": 0.392 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0536 + "score": 0.0431 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2492 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4081 + "score": 0.412 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1665 + "score": 0.1666 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3278 + "score": 0.3102 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.392 + "score": 0.3881 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0431 + "score": 0.0536 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2492 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.412 + "score": 0.4081 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1666 + "score": 0.1665 } } ], diff --git a/data/models/cpayne1303_llama-43m-beta.json b/data/models/cpayne1303_llama-43m-beta.json index 0f0d3430f35b22a1aef8ac071591532d594d6d4d..8e62d1be9edc1ccaeda7702d1a793fd5db4d4150 100644 --- a/data/models/cpayne1303_llama-43m-beta.json +++ b/data/models/cpayne1303_llama-43m-beta.json @@ -5,7 +5,7 @@ "developer": "cpayne1303", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "0.043" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1949 + "score": 0.1916 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2965 + "score": 0.2977 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0045 + "score": 0.0 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3885 + "score": 0.3872 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1111 + "score": 0.1132 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1916 + "score": 0.1949 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2965 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0045 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3872 + "score": 0.3885 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1132 + "score": 0.1111 } } ], diff --git a/data/models/daemontatox_documentcogito.json b/data/models/daemontatox_documentcogito.json index 157a2466b6d71022e4fcfd8ace48804c5b301a99..6383820e3f19a97d2a08bca3546e5cca52a154e3 100644 --- a/data/models/daemontatox_documentcogito.json +++ b/data/models/daemontatox_documentcogito.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MllamaForConditionalGeneration", "params_billions": "10.67" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5064 + "score": 0.777 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5112 + "score": 0.5187 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1631 + "score": 0.2198 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3163 + "score": 0.2936 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3973 + "score": 0.3911 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3738 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.777 + "score": 0.5064 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5187 + "score": 0.5112 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2198 + "score": 0.1631 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2936 + "score": 0.3163 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3911 + "score": 0.3973 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3738 + "score": 0.3802 } } ], diff --git a/data/models/daemontatox_pathfinderai.json b/data/models/daemontatox_pathfinderai.json index 8b13e2aabf79ff36b82f8d8bd4c0bcdf2b41e385..7a5f7d25c7278e2df08548a48abfe0b0ee9b4f2a 100644 --- a/data/models/daemontatox_pathfinderai.json +++ b/data/models/daemontatox_pathfinderai.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3745 + "score": 0.4855 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6668 + "score": 0.6627 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4841 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.3096 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4858 + "score": 0.4256 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5593 + "score": 0.5542 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4855 + "score": 0.3745 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6627 + "score": 0.6668 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4841 + "score": 0.4758 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3943 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4256 + "score": 0.4858 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5542 + "score": 0.5593 } } ], diff --git a/data/models/deepmount00_llama-3.1-8b-ita.json b/data/models/deepmount00_llama-3.1-8b-ita.json index be94466036e753700b8de72e15b468fc6edebdee..1fef7ca0b3d471692e7379016093b45312ecc7df 100644 --- a/data/models/deepmount00_llama-3.1-8b-ita.json +++ b/data/models/deepmount00_llama-3.1-8b-ita.json @@ -6,8 +6,8 @@ "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", - "architecture": "Unknown", - "params_billions": "0.0", + "architecture": "LlamaForCausalLM", + "params_billions": "8.03", "model_id_aliases": [ "DeepMount00/Llama-3.1-8b-Ita" ] @@ -15,7 +15,7 @@ }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -47,7 +47,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5365 + "score": 0.7917 } }, { @@ -65,7 +65,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.517 + "score": 0.5109 } }, { @@ -83,7 +83,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1707 + "score": 0.1088 } }, { @@ -101,7 +101,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.2878 } }, { @@ -119,7 +119,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4487 + "score": 0.4136 } }, { @@ -137,7 +137,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.396 + "score": 0.3876 } } ], @@ -145,7 +145,7 @@ "generation_config": null }, { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -177,7 +177,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7917 + "score": 0.5365 } }, { @@ -195,7 +195,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5109 + "score": 0.517 } }, { @@ -213,7 +213,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1088 + "score": 0.1707 } }, { @@ -231,7 +231,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2878 + "score": 0.3062 } }, { @@ -249,7 +249,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4136 + "score": 0.4487 } }, { @@ -267,7 +267,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3876 + "score": 0.396 } } ], diff --git a/data/models/deepseek_deepseek-r1-0528.json b/data/models/deepseek_deepseek-r1-0528.json index 22f7f1e7159f38877f415a9a6f07905fabcb5d33..712c8d82096fdfa393951ac4421e5d6c18ee722d 100644 --- a/data/models/deepseek_deepseek-r1-0528.json +++ b/data/models/deepseek_deepseek-r1-0528.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/deepseek_deepseek-v3.1.json b/data/models/deepseek_deepseek-v3.1.json index 28271d97eb42a16e44db35269bdfee0d33d2475a..52e144fa11c26943b339cba2c9fa28b1d9098c61 100644 --- a/data/models/deepseek_deepseek-v3.1.json +++ b/data/models/deepseek_deepseek-v3.1.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/doppelreflex_mn-12b-lilithframe.json b/data/models/doppelreflex_mn-12b-lilithframe.json index 2532c65fe82e1d06f8d68b438dbffc94e3e4e4bb..720fbe0846130d5605f8d9f0e7d4c1731ec8e47c 100644 --- a/data/models/doppelreflex_mn-12b-lilithframe.json +++ b/data/models/doppelreflex_mn-12b-lilithframe.json @@ -5,7 +5,7 @@ "developer": "DoppelReflEx", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "12.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.436 + "score": 0.451 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4956 + "score": 0.4944 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0589 + "score": 0.1156 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3205 + "score": 0.3196 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3843 + "score": 0.3896 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3237 + "score": 0.3256 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.451 + "score": 0.436 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4944 + "score": 0.4956 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1156 + "score": 0.0589 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3196 + "score": 0.3205 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3896 + "score": 0.3843 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3256 + "score": 0.3237 } } ], diff --git a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json index 34df651aabdc1d7685c9419a18517c0c9ee2b5a2..623bd9c2c8bfaa513dd516b6dc8551699ef2135b 100644 --- a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json +++ b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json @@ -5,7 +5,7 @@ "developer": "EpistemeAI", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7305 + "score": 0.7207 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4649 + "score": 0.461 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1397 + "score": 0.1314 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3209 + "score": 0.3432 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.348 + "score": 0.3354 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7207 + "score": 0.7305 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.461 + "score": 0.4649 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1314 + "score": 0.1397 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3432 + "score": 0.3209 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.348 } } ], diff --git a/data/models/etherll_herplete-llm-llama-3.1-8b.json b/data/models/etherll_herplete-llm-llama-3.1-8b.json index 31e2291931d31d5d1deeb312e00cc5de025980fd..1f50fb1261d1ec9c67c961ac513392f51c62e5ca 100644 --- a/data/models/etherll_herplete-llm-llama-3.1-8b.json +++ b/data/models/etherll_herplete-llm-llama-3.1-8b.json @@ -5,7 +5,7 @@ "developer": "Etherll", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6106 + "score": 0.4672 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.5013 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1548 + "score": 0.0279 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3146 + "score": 0.2861 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3991 + "score": 0.386 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3752 + "score": 0.3482 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4672 + "score": 0.6106 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5013 + "score": 0.5347 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0279 + "score": 0.1548 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3146 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3991 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3482 + "score": 0.3752 } } ], diff --git a/data/models/fblgit_thebeagle-v2beta-32b-mgs.json b/data/models/fblgit_thebeagle-v2beta-32b-mgs.json index 48527656a7b3e346612bd66367de25eb0d8e37d9..9045bce2e8448418fb4689f76787fc39b7a5722c 100644 --- a/data/models/fblgit_thebeagle-v2beta-32b-mgs.json +++ b/data/models/fblgit_thebeagle-v2beta-32b-mgs.json @@ -5,7 +5,7 @@ "developer": "fblgit", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4503 + "score": 0.5181 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7035 + "score": 0.7033 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.4947 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.401 + "score": 0.3826 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5021 + "score": 0.5008 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5911 + "score": 0.5915 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5181 + "score": 0.4503 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7033 + "score": 0.7035 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4947 + "score": 0.3943 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3826 + "score": 0.401 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5008 + "score": 0.5021 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5915 + "score": 0.5911 } } ], diff --git a/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json b/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json index 37f82f81803012f4270da262872645abffc35b8f..4037d46b51c1d5f1b2d23453c409340e1c080518 100644 --- a/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json +++ b/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7628 + "score": 0.7598 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5107 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.4237 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2802 + "score": 0.2768 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4579 + "score": 0.4539 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4033 + "score": 0.4012 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7598 + "score": 0.7628 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5107 + "score": 0.5098 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4237 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2802 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4539 + "score": 0.4579 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4012 + "score": 0.4033 } } ], diff --git a/data/models/google_gemini-2.5-flash.json b/data/models/google_gemini-2.5-flash.json index e86caee99d92f5b1111f5d4cfe31c4107b3564b0..820c8d9942dde2ebcfddb6efd01e86ba89a1eddf 100644 --- a/data/models/google_gemini-2.5-flash.json +++ b/data/models/google_gemini-2.5-flash.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.9, + "score": 17.1, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,7 +1376,7 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 16.9, "uncertainty": { "standard_error": { "value": 2.4 @@ -1386,7 +1386,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 15.4, + "score": 16.4, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1491,7 +1491,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1515,7 +1515,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1524,17 +1524,17 @@ "max_score": 100.0 }, "score_details": { - "score": 17.1, + "score": 15.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1551,7 +1551,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-2.5-pro.json b/data/models/google_gemini-2.5-pro.json index f2d7e76912c502afd2dae0560e9a20a114690132..ad69bc6f71351eff92acf181b72bb1ef3d2ca29a 100644 --- a/data/models/google_gemini-2.5-pro.json +++ b/data/models/google_gemini-2.5-pro.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.1, + "score": 32.6, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 26.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 32.6, + "score": 19.6, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1491,7 +1491,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1515,7 +1515,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1524,17 +1524,17 @@ "max_score": 100.0 }, "score_details": { - "score": 19.6, + "score": 16.4, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1551,7 +1551,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-flash.json b/data/models/google_gemini-3-flash.json index 8ad1fcf1488cbfb100a465260bd9105159bfdda1..47fa2df5c9baf77e090796d7a217aa9e4577803b 100644 --- a/data/models/google_gemini-3-flash.json +++ b/data/models/google_gemini-3-flash.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { - "agent_name": "Junie CLI", - "agent_organization": "JetBrains" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.3, + "score": 51.7, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-07", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7, + "score": 64.3, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-pro-preview.json b/data/models/google_gemini-3-pro-preview.json index 3f20467a04db22c43c2d253a7e0a111a72926152..003d3891c3ed687a312bcf65a5ce96e8186ce448 100644 --- a/data/models/google_gemini-3-pro-preview.json +++ b/data/models/google_gemini-3-pro-preview.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro-preview", "developer": "Google", "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.13, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "130.49", - "average_steps": "22.59", - "percent_finished": "1.0" + "average_agent_cost": "2.54", + "total_run_cost": "254.25", + "average_steps": "49.13", + "percent_finished": "0.71" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -70,8 +70,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -146,7 +146,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -159,42 +159,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.505, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.88", - "total_run_cost": "188.19", - "average_steps": "21.76", - "percent_finished": "0.99" + "average_agent_cost": "2.85", + "total_run_cost": "284.68", + "average_steps": "22.88", + "percent_finished": "0.7" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -246,23 +246,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.36, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.11", - "total_run_cost": "310.55", - "average_steps": "38.01", - "percent_finished": "0.86" + "average_agent_cost": "1.3", + "total_run_cost": "130.49", + "average_steps": "22.59", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3333, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.64", - "total_run_cost": "63.79", - "average_steps": "8.45", - "percent_finished": "0.6061" + "average_agent_cost": "2.39", + "total_run_cost": "239.0", + "average_steps": "29.63", + "percent_finished": "0.69" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -363,42 +363,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.505, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.85", - "total_run_cost": "284.68", - "average_steps": "22.88", - "percent_finished": "0.7" + "average_agent_cost": "1.88", + "total_run_cost": "188.19", + "average_steps": "21.76", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.13, + "score": 0.36, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.54", - "total_run_cost": "254.25", - "average_steps": "49.13", - "percent_finished": "0.71" + "average_agent_cost": "3.11", + "total_run_cost": "310.55", + "average_steps": "38.01", + "percent_finished": "0.86" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.3333, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.44", - "total_run_cost": "44.18", - "average_steps": "7.85", - "percent_finished": "0.99" + "average_agent_cost": "0.64", + "total_run_cost": "63.79", + "average_steps": "8.45", + "percent_finished": "0.6061" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.39", - "total_run_cost": "239.0", - "average_steps": "29.63", - "percent_finished": "0.69" + "average_agent_cost": "0.44", + "total_run_cost": "44.18", + "average_steps": "7.85", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1856,7 +1856,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1869,33 +1869,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.62, + "score": 0.71, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.18", - "average_steps": "10.9", + "average_agent_cost": "0.7", + "total_run_cost": "69.56", + "average_steps": "32.55", "percent_finished": "1.0" } }, @@ -1903,8 +1903,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1916,15 +1916,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1971,8 +1971,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1984,15 +1984,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2024,14 +2024,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.7234, "uncertainty": { - "num_samples": 100 + "num_samples": 94 }, "details": { - "average_agent_cost": "0.7", - "total_run_cost": "69.56", - "average_steps": "32.55", + "average_agent_cost": "1.58", + "total_run_cost": "148.44", + "average_steps": "32.36", "percent_finished": "1.0" } }, @@ -2039,8 +2039,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2052,15 +2052,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2073,33 +2073,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7234, + "score": 0.7, "uncertainty": { - "num_samples": 94 + "num_samples": 50 }, "details": { - "average_agent_cost": "1.58", - "total_run_cost": "148.44", - "average_steps": "32.36", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2107,8 +2107,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2120,15 +2120,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2160,14 +2160,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.62, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.34", - "total_run_cost": "17.45", - "average_steps": "12.62", + "average_agent_cost": "0.21", + "total_run_cost": "11.18", + "average_steps": "10.9", "percent_finished": "1.0" } }, @@ -2175,8 +2175,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2188,15 +2188,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2233,9 +2233,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.34", + "total_run_cost": "17.45", + "average_steps": "12.62", "percent_finished": "1.0" } }, @@ -2243,8 +2243,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2256,15 +2256,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2296,14 +2296,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.2", - "total_run_cost": "10.29", - "average_steps": "12.28", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2311,8 +2311,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2324,15 +2324,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2364,14 +2364,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.68, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.2", + "total_run_cost": "10.29", + "average_steps": "12.28", "percent_finished": "1.0" } }, @@ -2379,8 +2379,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2392,15 +2392,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2432,14 +2432,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.82, + "score": 0.7576, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "16.64", - "average_steps": "11.25", + "average_agent_cost": "0.21", + "total_run_cost": "21.43", + "average_steps": "11.3", "percent_finished": "1.0" } }, @@ -2447,8 +2447,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2460,15 +2460,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2500,14 +2500,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.7805, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.27", - "total_run_cost": "27.48", - "average_steps": "10.62", + "average_agent_cost": "0.19", + "total_run_cost": "19.38", + "average_steps": "11.18", "percent_finished": "1.0" } }, @@ -2515,8 +2515,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2528,15 +2528,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2583,8 +2583,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2596,15 +2596,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2636,14 +2636,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "21.43", - "average_steps": "11.3", + "average_agent_cost": "0.27", + "total_run_cost": "27.48", + "average_steps": "10.62", "percent_finished": "1.0" } }, @@ -2651,8 +2651,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2664,15 +2664,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2685,33 +2685,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_telecom", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/telecom", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/telecom", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.82, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.16", + "total_run_cost": "16.64", + "average_steps": "11.25", "percent_finished": "1.0" } }, @@ -2740,7 +2740,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2772,23 +2772,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8876, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.54", - "total_run_cost": "58.29", - "average_steps": "10.82", - "percent_finished": "0.89" + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2800,15 +2800,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2840,23 +2840,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.88, + "score": 0.8876, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.35", - "total_run_cost": "40.25", - "average_steps": "12.71", - "percent_finished": "1.0" + "average_agent_cost": "0.54", + "total_run_cost": "58.29", + "average_steps": "10.82", + "percent_finished": "0.89" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2868,15 +2868,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2889,33 +2889,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_telecom", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/telecom", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/telecom", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7805, + "score": 0.6852, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.19", - "total_run_cost": "19.38", - "average_steps": "11.18", + "average_agent_cost": "0.21", + "total_run_cost": "25.48", + "average_steps": "9.9", "percent_finished": "1.0" } }, @@ -2944,7 +2944,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2976,14 +2976,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.88, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.35", + "total_run_cost": "40.25", + "average_steps": "12.71", "percent_finished": "1.0" } }, @@ -2991,8 +2991,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -3004,15 +3004,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -3044,14 +3044,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6852, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "25.48", - "average_steps": "9.9", + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", "percent_finished": "1.0" } }, @@ -3059,8 +3059,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -3072,8 +3072,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } diff --git a/data/models/google_gemini-3-pro.json b/data/models/google_gemini-3-pro.json index 5acfd6f3e4c0e364f178f7fde5f99550b7d69ddb..11bc2c604594dba4ea7261f6d2e8fc51b386c982 100644 --- a/data/models/google_gemini-3-pro.json +++ b/data/models/google_gemini-3-pro.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro", "developer": "Google", "additional_details": { - "agent_name": "Droid", - "agent_organization": "Factory" + "agent_name": "SageAgent", + "agent_organization": "OpenSage" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/sageagent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.1, + "score": 65.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/sageagent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 65.2, + "score": 61.1, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"SageAgent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-21", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.9, + "score": 56.0, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-01-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.8, + "score": 69.4, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.0, + "score": 61.8, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-06", + "evaluation_timestamp": "2025-11-21", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 69.4, + "score": 56.9, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3.1-pro.json b/data/models/google_gemini-3.1-pro.json index d5a56163d0a60a102dbdbef6cd39c4565d8d4e26..ecc327e910f8e4125f6002db506fbf114eb7832a 100644 --- a/data/models/google_gemini-3.1-pro.json +++ b/data/models/google_gemini-3.1-pro.json @@ -4,13 +4,13 @@ "id": "google/gemini-3.1-pro", "developer": "Google", "additional_details": { - "agent_name": "Terminus-KIRA", - "agent_organization": "KRAFTON AI" + "agent_name": "Forge Code", + "agent_organization": "Forge Code" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-03-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.8, + "score": 78.4, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 1.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-02", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 78.4, + "score": 74.8, "uncertainty": { "standard_error": { - "value": 1.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini_3_flash.json b/data/models/google_gemini_3_flash.json index 6beb736773081d366f65edf05d1bf54bd7e07ce2..3a911939d887cdc9433884fdde5ee11c97c277a4 100644 --- a/data/models/google_gemini_3_flash.json +++ b/data/models/google_gemini_3_flash.json @@ -6,6 +6,53 @@ "inference_platform": "unknown" }, "evaluations": [ + { + "evaluation_id": "ace/google_gemini-3-flash/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.415 + }, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + }, { "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", @@ -205,53 +252,6 @@ } } }, - { - "evaluation_id": "ace/google_gemini-3-flash/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.415 - }, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - }, { "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/google_gemma-3-27b-it.json b/data/models/google_gemma-3-27b-it.json index 0d22aa7a55f613493f23d93430a44590b7aa715d..31e90c4548397bec1dec70a558a4830dc0c4f7c9 100644 --- a/data/models/google_gemma-3-27b-it.json +++ b/data/models/google_gemma-3-27b-it.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/gunulhona_gemma-ko-merge-peft.json b/data/models/gunulhona_gemma-ko-merge-peft.json index 632db743d31cc31d893a565c873ea7be7cc73fb2..7aaca8e76ac4d54a166ddba85afcaa38d76748c4 100644 --- a/data/models/gunulhona_gemma-ko-merge-peft.json +++ b/data/models/gunulhona_gemma-ko-merge-peft.json @@ -5,7 +5,7 @@ "developer": "Gunulhona", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "?", "params_billions": "20.318" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.288 + "score": 0.4441 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5154 + "score": 0.4863 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.307 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.408 + "score": 0.3986 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3817 + "score": 0.3098 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4441 + "score": 0.288 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4863 + "score": 0.5154 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3247 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3986 + "score": 0.408 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3098 + "score": 0.3817 } } ], diff --git a/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json b/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json index 357f438b07831c53712dd63f870b6b2401c4d681..69bdc21b5b14ca68b6163472e8eef6fd89e268f5 100644 --- a/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json +++ b/data/models/hendrydong_mistral-rm-for-raft-gshf-v0.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5851 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5779 + "score": 0.7847 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6011 + "score": 0.5789 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6956 + "score": 0.85 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6747 + "score": 0.7434 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5988 + "score": 0.7508 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7847 + "score": 0.5851 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5779 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5789 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6011 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.85 + "score": 0.6956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7434 + "score": 0.6747 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7508 + "score": 0.5988 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/huggingfacetb_smollm2-360m-instruct.json b/data/models/huggingfacetb_smollm2-360m-instruct.json index a4edda6b87f3cb7e0eeabaee9333bba91d019d12..b9a479d91c3b9226f2e42a0a5392df69dc01d9df 100644 --- a/data/models/huggingfacetb_smollm2-360m-instruct.json +++ b/data/models/huggingfacetb_smollm2-360m-instruct.json @@ -5,9 +5,9 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "0.362" + "params_billions": "0.36" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.083 + "score": 0.3842 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3053 + "score": 0.3144 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0083 + "score": 0.0151 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2651 + "score": 0.255 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3423 + "score": 0.3461 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1117 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.083 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3144 + "score": 0.3053 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0151 + "score": 0.0083 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.255 + "score": 0.2651 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3461 + "score": 0.3423 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1117 + "score": 0.1126 } } ], diff --git a/data/models/infly_inf-orm-llama3.1-70b.json b/data/models/infly_inf-orm-llama3.1-70b.json index 82e76ad6cd43b3a105b801967d5f616a3924844a..e7947ee940015eb0652da9a52891a9ab47739595 100644 --- a/data/models/infly_inf-orm-llama3.1-70b.json +++ b/data/models/infly_inf-orm-llama3.1-70b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9511 + "score": 0.7648 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.7411 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9101 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9365 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9912 + "score": 0.9644 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7648 + "score": 0.903 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7411 + "score": 0.8622 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9511 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9644 + "score": 0.9101 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.903 + "score": 0.9365 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8622 + "score": 0.9912 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/internlm_internlm2-1_8b-reward.json b/data/models/internlm_internlm2-1_8b-reward.json index db02a96edd7d4deec66c9db68ed23c5ecfdc96f9..fdd95af043dae0ccf6943006b6d82bc9a69847ba 100644 --- a/data/models/internlm_internlm2-1_8b-reward.json +++ b/data/models/internlm_internlm2-1_8b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8217 + "score": 0.3902 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9358 + "score": 0.2758 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6623 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.4426 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8724 + "score": 0.4711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3902 + "score": 0.596 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2758 + "score": 0.1934 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.8217 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4426 + "score": 0.9358 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4711 + "score": 0.6623 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.596 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1934 + "score": 0.8724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/internlm_internlm2-20b-reward.json b/data/models/internlm_internlm2-20b-reward.json index db57bc6ddd293d585b6bca7ac06b0b270dabb864..4de166855af75438b275f87369424e25bd91920e 100644 --- a/data/models/internlm_internlm2-20b-reward.json +++ b/data/models/internlm_internlm2-20b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", + "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5628 + "score": 0.9016 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5558 + "score": 0.9888 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.7654 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5738 + "score": 0.8946 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6111 + "score": 0.9576 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7253 + "score": 0.5628 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5483 + "score": 0.5558 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9016 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9888 + "score": 0.5738 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7654 + "score": 0.6111 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8946 + "score": 0.7253 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9576 + "score": 0.5483 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/isaak-carter_josiev4o-8b-stage1-v4.json b/data/models/isaak-carter_josiev4o-8b-stage1-v4.json index 4be5b49897aa6863276434fc48c4d6962992c15e..3d8b7bc1575ec467cf84b27d25552c6269300054 100644 --- a/data/models/isaak-carter_josiev4o-8b-stage1-v4.json +++ b/data/models/isaak-carter_josiev4o-8b-stage1-v4.json @@ -5,7 +5,7 @@ "developer": "Isaak-Carter", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2553 + "score": 0.2477 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4725 + "score": 0.4758 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0529 + "score": 0.0453 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2919 + "score": 0.2911 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3654 + "score": 0.3641 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3316 + "score": 0.3292 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2477 + "score": 0.2553 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4725 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0453 + "score": 0.0529 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2911 + "score": 0.2919 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3641 + "score": 0.3654 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3316 } } ], diff --git a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json index 88cf241ec28671fc887627a1e7bb1af09c15f9cf..1160e36435338576bea29bef2eb39f295798e22d 100644 --- a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json +++ b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json @@ -5,7 +5,7 @@ "developer": "jaspionjader", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4418 + "score": 0.4345 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.5419 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1352 + "score": 0.1292 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3087 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3854 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4345 + "score": 0.4418 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5419 + "score": 0.5406 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1292 + "score": 0.1352 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3062 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3854 + "score": 0.386 } } ], diff --git a/data/models/kavonalds_bundermaxx-0710.json b/data/models/kavonalds_bundermaxx-0710.json index 1c392120900a40ecef63990c2ea6f8c3e297a9f1..c6ac0f8a7475b90498abe064315ec79f3f2bd920 100644 --- a/data/models/kavonalds_bundermaxx-0710.json +++ b/data/models/kavonalds_bundermaxx-0710.json @@ -5,7 +5,7 @@ "developer": "kavonalds", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3283 + "score": 0.2701 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6651 + "score": 0.5566 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2802 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3393 + "score": 0.3682 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1314 + "score": 0.1449 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.3283 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5566 + "score": 0.6651 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2802 + "score": 0.2609 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3682 + "score": 0.3393 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1449 + "score": 0.1314 } } ], diff --git a/data/models/lxzgordon_urm-llama-3.1-8b.json b/data/models/lxzgordon_urm-llama-3.1-8b.json index 7f03035f2bd809fb14be27130e313f5b86a26f9a..2ce56c90ce0d5fb5788660f5b3f1f1e179701786 100644 --- a/data/models/lxzgordon_urm-llama-3.1-8b.json +++ b/data/models/lxzgordon_urm-llama-3.1-8b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7394 + "score": 0.9294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6884 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.45 + "score": 0.8816 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6393 + "score": 0.9108 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9178 + "score": 0.9698 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9758 + "score": 0.7394 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7653 + "score": 0.6884 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9294 + "score": 0.45 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.6393 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8816 + "score": 0.9178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9108 + "score": 0.9758 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9698 + "score": 0.7653 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json index 2ac3bb48cceaf59e0da754702b99c90edec98998..018a3210e6da83ca7e6d3501fd6a7fefdb10449c 100644 --- a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json +++ b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json @@ -5,7 +5,7 @@ "developer": "Magpie-Align", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4027 + "score": 0.4118 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.4811 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0461 + "score": 0.034 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2752 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3047 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3001 + "score": 0.3006 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4118 + "score": 0.4027 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4811 + "score": 0.4789 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.034 + "score": 0.0461 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.2768 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3047 + "score": 0.3087 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3006 + "score": 0.3001 } } ], diff --git a/data/models/meta-llama_meta-llama-3-8b-instruct.json b/data/models/meta-llama_meta-llama-3-8b-instruct.json index 7f250b5cb54b80c03e9c7605296b8768c6176139..6b360a3e795c08b36f8a5ae66db62ef8475fd282 100644 --- a/data/models/meta-llama_meta-llama-3-8b-instruct.json +++ b/data/models/meta-llama_meta-llama-3-8b-instruct.json @@ -5,7 +5,7 @@ "developer": "meta-llama", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7408 + "score": 0.4782 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4989 + "score": 0.491 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0869 + "score": 0.0914 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2928 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3568 + "score": 0.3805 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3664 + "score": 0.3591 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4782 + "score": 0.7408 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.491 + "score": 0.4989 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0914 + "score": 0.0869 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2928 + "score": 0.2592 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3568 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3591 + "score": 0.3664 } } ], diff --git a/data/models/mistralai_mistral-small-2503.json b/data/models/mistralai_mistral-small-2503.json index be5d73de7278abb3c747dbebd44b60d3fa624503..6df0d972b005ae32b060fdc4673d6092e770670f 100644 --- a/data/models/mistralai_mistral-small-2503.json +++ b/data/models/mistralai_mistral-small-2503.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/mistralai_mistral-small-instruct-2409.json b/data/models/mistralai_mistral-small-instruct-2409.json index 02a04ce41b1de063befb30b7d94d3ad197cffb0e..5aac65e94f46e67333b7cb66dd77ae7f5415d17b 100644 --- a/data/models/mistralai_mistral-small-instruct-2409.json +++ b/data/models/mistralai_mistral-small-instruct-2409.json @@ -5,9 +5,9 @@ "developer": "mistralai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", - "params_billions": "22.05" + "params_billions": "22.247" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.667 + "score": 0.6283 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5213 + "score": 0.583 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1435 + "score": 0.2039 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3331 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3632 + "score": 0.4063 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.396 + "score": 0.4099 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6283 + "score": 0.667 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.583 + "score": 0.5213 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2039 + "score": 0.1435 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3331 + "score": 0.3238 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4063 + "score": 0.3632 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.396 } } ], diff --git a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json index 7ef165972eafdeec56d923c82e02fdbbc9479eac..d443de39bb7ed82b00df80190432e583c21fd660 100644 --- a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json +++ b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json @@ -5,7 +5,7 @@ "developer": "mlabonne", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.7561 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5124 + "score": 0.5111 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0853 + "score": 0.0906 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3029 + "score": 0.3062 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.415 + "score": 0.4019 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3841 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7561 + "score": 0.4162 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5111 + "score": 0.5124 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0906 + "score": 0.0853 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3029 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.415 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3841 + "score": 0.3802 } } ], diff --git a/data/models/moonshot-ai_kimi-k2-instruct.json b/data/models/moonshot-ai_kimi-k2-instruct.json index 2cfded0145e5c6821159f45b392f6b86e15c7f49..758984500ae56b028445a56fb8562c74a90a3a8f 100644 --- a/data/models/moonshot-ai_kimi-k2-instruct.json +++ b/data/models/moonshot-ai_kimi-k2-instruct.json @@ -4,13 +4,13 @@ "id": "moonshot-ai/kimi-k2-instruct", "developer": "Moonshot AI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.8, + "score": 26.7, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.7, + "score": 27.8, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/multiple_multiple.json b/data/models/multiple_multiple.json index 01b06a48bec6c0ec600ef83ad482f6552d46d980..45d5496f95937f9e2416eacde936df0b731a4892 100644 --- a/data/models/multiple_multiple.json +++ b/data/models/multiple_multiple.json @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-11-20", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,10 +43,10 @@ "max_score": 100.0 }, "score_details": { - "score": 61.2, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-20", + "evaluation_timestamp": "2026-03-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 71.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-05", + "evaluation_timestamp": "2025-11-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 72.4, + "score": 50.1, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-11", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 50.1, + "score": 58.4, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-07", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.0, + "score": 61.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-03-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4, + "score": 72.4, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/ncsoft_llama-3-offsetbias-rm-8b.json b/data/models/ncsoft_llama-3-offsetbias-rm-8b.json index 14fabae93f5ca8f04e2a0c2d895536cd5cdfcb3a..0a839fbe35d3881341976e6cd6ecdba200fd0827 100644 --- a/data/models/ncsoft_llama-3-offsetbias-rm-8b.json +++ b/data/models/ncsoft_llama-3-offsetbias-rm-8b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", + "evaluation_id": "reward-bench/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.648 + "score": 0.8942 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6084 + "score": 0.9721 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4 + "score": 0.818 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7222 + "score": 0.9192 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9596 + "score": 0.648 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6786 + "score": 0.6084 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8942 + "score": 0.4 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9721 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.818 + "score": 0.7222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.9596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9192 + "score": 0.6786 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/nexusflow_starling-rm-34b.json b/data/models/nexusflow_starling-rm-34b.json index 0373bd963cc9795a2ad38fc9da30f8417d64ff6d..8ab4f392fbf3047e7bcb88c6ed9f3a2b6d8e5a37 100644 --- a/data/models/nexusflow_starling-rm-34b.json +++ b/data/models/nexusflow_starling-rm-34b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8133 + "score": 0.4553 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.4589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3187 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6175 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.877 + "score": 0.7556 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8845 + "score": 0.4808 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7137 + "score": 0.1004 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4553 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4589 + "score": 0.8133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3187 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6175 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7556 + "score": 0.877 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4808 + "score": 0.8845 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1004 + "score": 0.7137 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json index 486ba5e61208261c68f73d7d2bf88d78b1b36131..f86ab61b521749eb0f8fa030a3c29f5a32d327a1 100644 --- a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json +++ b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json @@ -5,7 +5,7 @@ "developer": "ontocord", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.759" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1162 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3171 + "score": 0.3184 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0113 + "score": 0.0076 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2685 + "score": 0.2634 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.3447 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1124 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1162 + "score": 0.1128 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3184 + "score": 0.3171 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0113 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2634 + "score": 0.2685 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3447 + "score": 0.346 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1124 + "score": 0.1129 } } ], diff --git a/data/models/openai_gpt-4o-2024-08-06.json b/data/models/openai_gpt-4o-2024-08-06.json index ca15abfce433f9cf3b132cc5fefadfcef719867a..4523783fb76c33e56985cf8706dfcde93c76d009 100644 --- a/data/models/openai_gpt-4o-2024-08-06.json +++ b/data/models/openai_gpt-4o-2024-08-06.json @@ -1900,10 +1900,10 @@ } }, { - "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", + "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1922,104 +1922,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6493 + "score": 0.8673 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5684 + "score": 0.9609 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.623 + "score": 0.8811 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8619 + "score": 0.8661 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.6493 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2028,135 +2052,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7819 + "score": 0.5684 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8673 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9609 + "score": 0.623 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.8619 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8811 + "score": 0.7293 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8661 + "score": 0.7819 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openai_gpt-4o-mini-2024-07-18.json b/data/models/openai_gpt-4o-mini-2024-07-18.json index 1b3fb4c30102ee1f603e3640bb4c4b14c38b8cac..f38f4e1ddedf08dfff3e3ceeb97aeebb3dcce913 100644 --- a/data/models/openai_gpt-4o-mini-2024-07-18.json +++ b/data/models/openai_gpt-4o-mini-2024-07-18.json @@ -2124,10 +2124,10 @@ } }, { - "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2146,104 +2146,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5796 + "score": 0.8007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4105 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.6075 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.8081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8374 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7414 + "score": 0.5796 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2252,135 +2276,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6962 + "score": 0.4105 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8007 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6075 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8081 + "score": 0.7414 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8374 + "score": 0.6962 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openai_gpt-5-2025-08-07.json b/data/models/openai_gpt-5-2025-08-07.json index fdb97ce4111978e08938be252c1bd72645c11969..0853492fcc4bbd45f07185718454686350fe1be2 100644 --- a/data/models/openai_gpt-5-2025-08-07.json +++ b/data/models/openai_gpt-5-2025-08-07.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/openai_gpt-5-codex.json b/data/models/openai_gpt-5-codex.json index bfb9d809d90faf4023c720a69d517cc43ddb6092..21a2358b27ce78cb6bd659dfe0536dac012c1818 100644 --- a/data/models/openai_gpt-5-codex.json +++ b/data/models/openai_gpt-5-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.4, + "score": 41.3, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 41.3, + "score": 43.4, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-mini.json b/data/models/openai_gpt-5-mini.json index c2da3f31323642732ef7d6d76ecea92f3ae2ea0b..a9c814071acbfdd08e7ec7cb2178b705e66b8da6 100644 --- a/data/models/openai_gpt-5-mini.json +++ b/data/models/openai_gpt-5-mini.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-mini", "developer": "OpenAI", "additional_details": { - "agent_name": "Codex CLI", - "agent_organization": "OpenAI" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 31.9, + "score": 29.2, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 31.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 22.2, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 22.2, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-nano.json b/data/models/openai_gpt-5-nano.json index 17031b27e97ab53bd889b020db9c4f3335b70c14..0d64dfdab152e442151bbf86365c72d39472c605 100644 --- a/data/models/openai_gpt-5-nano.json +++ b/data/models/openai_gpt-5-nano.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-nano", "developer": "OpenAI", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 7.0, + "score": 9.9, "uncertainty": { "standard_error": { - "value": 1.9 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 9.9, + "score": 11.5, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 11.5, + "score": 7.9, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 1.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.9, + "score": 7.0, "uncertainty": { "standard_error": { "value": 1.9 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.1-codex.json b/data/models/openai_gpt-5.1-codex.json index d4b36a19fcc7086223ae8f218d54f2b1a6ca0184..5ac777453649e155d2cf4ca5c0edb4c1fb8d16a4 100644 --- a/data/models/openai_gpt-5.1-codex.json +++ b/data/models/openai_gpt-5.1-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.1-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Crux", - "agent_organization": "Roam" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-16", + "evaluation_timestamp": "2025-11-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-16", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 53.5, + "score": 57.8, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-17", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 53.5, "uncertainty": { "standard_error": { - "value": 3.2 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.2-2025-12-11.json b/data/models/openai_gpt-5.2-2025-12-11.json index 296a87eb83719a9fcbce40e624e36b0e1264dcb7..b1415af8e1fa1580e027129e48746b8a6ec316b5 100644 --- a/data/models/openai_gpt-5.2-2025-12-11.json +++ b/data/models/openai_gpt-5.2-2025-12-11.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2-2025-12-11", "developer": "OpenAI", "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.071, + "score": 0.22, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "55.03", - "average_steps": "51.59", - "percent_finished": "0.61" + "average_agent_cost": "0.36", + "total_run_cost": "36.37", + "average_steps": "10.05", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -125,8 +125,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.22, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.36", - "total_run_cost": "36.37", - "average_steps": "10.05", - "percent_finished": "1.0" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -261,8 +261,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.071, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.55", + "total_run_cost": "55.03", + "average_steps": "51.59", + "percent_finished": "0.61" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,23 +382,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.43, + "score": 0.46, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.43", - "total_run_cost": "43.11", - "average_steps": "8.97", - "percent_finished": "1.0" + "average_agent_cost": "0.3", + "total_run_cost": "29.78", + "average_steps": "8.14", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.26, + "score": 0.43, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.17", - "total_run_cost": "17.31", - "average_steps": "6.57", - "percent_finished": "0.99" + "average_agent_cost": "0.43", + "total_run_cost": "43.11", + "average_steps": "8.97", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.46, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "29.78", - "average_steps": "8.14", - "percent_finished": "0.99" + "average_agent_cost": "0.38", + "total_run_cost": "38.21", + "average_steps": "14.27", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,14 +586,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.46, + "score": 0.26, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "29.78", - "average_steps": "8.14", + "average_agent_cost": "0.17", + "total_run_cost": "17.31", + "average_steps": "6.57", "percent_finished": "0.99" } }, @@ -601,8 +601,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -654,23 +654,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.46, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.38", - "total_run_cost": "38.21", - "average_steps": "14.27", - "percent_finished": "1.0" + "average_agent_cost": "0.3", + "total_run_cost": "29.78", + "average_steps": "8.14", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -769,7 +769,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -801,14 +801,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5253, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.45", - "total_run_cost": "44.58", - "average_steps": "19.98", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -816,8 +816,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -829,15 +829,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -869,14 +869,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.94", + "total_run_cost": "93.98", + "average_steps": "23.99", "percent_finished": "1.0" } }, @@ -884,8 +884,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -897,15 +897,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -937,14 +937,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5455, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.26", + "total_run_cost": "25.64", + "average_steps": "20.44", "percent_finished": "1.0" } }, @@ -952,8 +952,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -965,15 +965,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1005,14 +1005,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.5253, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.94", - "total_run_cost": "93.98", - "average_steps": "23.99", + "average_agent_cost": "0.45", + "total_run_cost": "44.58", + "average_steps": "19.98", "percent_finished": "1.0" } }, @@ -1020,8 +1020,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1033,15 +1033,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1054,33 +1054,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.57, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -1177,7 +1177,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1209,14 +1209,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.29", - "total_run_cost": "15.28", - "average_steps": "10.68", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1224,8 +1224,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1237,15 +1237,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1277,14 +1277,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.48, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.21", + "total_run_cost": "11.23", + "average_steps": "10.18", "percent_finished": "1.0" } }, @@ -1292,8 +1292,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1305,15 +1305,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1326,33 +1326,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5455, + "score": 0.54, "uncertainty": { - "num_samples": 99 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.26", - "total_run_cost": "25.64", - "average_steps": "20.44", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1360,8 +1360,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1373,15 +1373,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1413,14 +1413,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.6, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.23", - "average_steps": "10.18", + "average_agent_cost": "0.29", + "total_run_cost": "15.28", + "average_steps": "10.68", "percent_finished": "1.0" } }, @@ -1428,8 +1428,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1441,15 +1441,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1481,23 +1481,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", - "percent_finished": "1.0" + "total_run_cost": "11.54", + "average_steps": "9.55", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1509,15 +1509,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1564,8 +1564,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1577,15 +1577,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1617,23 +1617,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "26.27", - "average_steps": "11.08", - "percent_finished": "1.0" + "average_agent_cost": "0.12", + "total_run_cost": "12.63", + "average_steps": "9.92", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1645,15 +1645,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1685,23 +1685,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "11.54", - "average_steps": "9.55", - "percent_finished": "0.99" + "average_agent_cost": "0.25", + "total_run_cost": "26.27", + "average_steps": "11.08", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1713,15 +1713,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1734,33 +1734,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_telecom", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/telecom", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/telecom", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "35.31", - "average_steps": "10.11", + "average_agent_cost": "0.11", + "total_run_cost": "12.27", + "average_steps": "10.33", "percent_finished": "1.0" } }, @@ -1768,8 +1768,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1781,15 +1781,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1821,23 +1821,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.1", - "total_run_cost": "15.15", - "average_steps": "9.36", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1849,8 +1849,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1993,7 +1993,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2025,23 +2025,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.71, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.3", + "total_run_cost": "35.31", + "average_steps": "10.11", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2053,15 +2053,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2074,34 +2074,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_telecom", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/telecom", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/telecom", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.12", - "total_run_cost": "12.63", - "average_steps": "9.92", - "percent_finished": "0.98" + "average_agent_cost": "0.1", + "total_run_cost": "15.15", + "average_steps": "9.36", + "percent_finished": "1.0" } }, "generation_config": { diff --git a/data/models/openai_gpt-5.2.json b/data/models/openai_gpt-5.2.json index ff63f2052ea0b7c9a5cea40f5696ce46050721df..77930b98006acfd2c6efa9a1e831cbd44874ee48 100644 --- a/data/models/openai_gpt-5.2.json +++ b/data/models/openai_gpt-5.2.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,11 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 60.7 + "score": 64.9, + "uncertainty": { + "standard_error": { + "value": 2.8 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -138,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -152,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -176,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -185,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.9, + "score": 54.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,17 +265,11 @@ "max_score": 100.0 }, "score_details": { - "score": 54.0, - "uncertainty": { - "standard_error": { - "value": 2.9 - }, - "num_samples": 435 - } + "score": 60.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.3-codex.json b/data/models/openai_gpt-5.3-codex.json index 65f8130b44c7c4318430a774c580739b4e8b1cab..29bf24de6a839acf45f0dd14a04022b9052f2277 100644 --- a/data/models/openai_gpt-5.3-codex.json +++ b/data/models/openai_gpt-5.3-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.3-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Mux", - "agent_organization": "Coder" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.6, + "score": 64.7, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.7, + "score": 75.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 75.1, + "score": 77.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-24", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 77.3, + "score": 74.6, "uncertainty": { "standard_error": { - "value": 2.2 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-oss-20b.json b/data/models/openai_gpt-oss-20b.json index 584e35a6df7897b2cd4e9e1cdb5c6db8d87d90ab..d658cd5464b4783e6e0372fda0cb20e4e0a6a422 100644 --- a/data/models/openai_gpt-oss-20b.json +++ b/data/models/openai_gpt-oss-20b.json @@ -310,7 +310,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-20b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-20b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -334,7 +334,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -343,17 +343,17 @@ "max_score": 100.0 }, "score_details": { - "score": 3.1, + "score": 3.4, "uncertainty": { "standard_error": { - "value": 1.5 + "value": 1.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-20B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-20B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -370,7 +370,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-20B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-20B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -384,7 +384,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-20b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-20b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -408,7 +408,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -417,17 +417,17 @@ "max_score": 100.0 }, "score_details": { - "score": 3.4, + "score": 3.1, "uncertainty": { "standard_error": { - "value": 1.4 + "value": 1.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-20B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-20B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -444,7 +444,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-20B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-20B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt_5.2.json b/data/models/openai_gpt_5.2.json index e5de01feca478b0842db2b2533d006446492021b..fec8e3d9659540e746cd368872cad0ec3c496416 100644 --- a/data/models/openai_gpt_5.2.json +++ b/data/models/openai_gpt_5.2.json @@ -7,10 +7,10 @@ }, "evaluations": [ { - "evaluation_id": "apex-agents/openai_gpt-5.2/1773260200", + "evaluation_id": "ace/openai_gpt-5.2/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { - "source_name": "Mercor APEX-Agents Leaderboard", + "source_name": "Mercor ACE Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", @@ -20,24 +20,24 @@ "name": "archipelago", "version": "1.0.0" }, - "benchmark": "apex-agents", + "benchmark": "ace", "evaluation_results": [ { - "evaluation_name": "Overall Pass@1", + "evaluation_name": "Overall Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", + "evaluation_description": "Overall ACE score across all consumer-task domains.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.23, + "score": 0.515, "uncertainty": { "confidence_interval": { "lower": -0.032, @@ -53,28 +53,21 @@ } }, { - "evaluation_name": "Overall Pass@8", + "evaluation_name": "Food Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Overall Pass@8 (dataset card / paper snapshot).", + "evaluation_description": "Food domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.4, - "uncertainty": { - "confidence_interval": { - "lower": -0.044, - "upper": 0.044, - "method": "bootstrap" - } - } + "score": 0.65 }, "generation_config": { "additional_details": { @@ -83,44 +76,75 @@ } }, { - "evaluation_name": "Overall Mean Score", + "evaluation_name": "Gaming Score", "source_data": { - "dataset_name": "apex-agents", + "dataset_name": "ace", "source_type": "hf_dataset", - "hf_repo": "mercor/apex-agents" + "hf_repo": "Mercor/ACE" }, "metric_config": { - "evaluation_description": "Overall mean rubric score.", + "evaluation_description": "Gaming domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.387 + "score": 0.578 }, "generation_config": { "additional_details": { "run_setting": "High" } } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + }, + { + "evaluation_id": "apex-agents/openai_gpt-5.2/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor APEX-Agents Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "apex-agents", + "evaluation_results": [ { - "evaluation_name": "Investment Banking Pass@1", + "evaluation_name": "Overall Pass@1", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Investment banking world Pass@1.", + "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.273 + "score": 0.23, + "uncertainty": { + "confidence_interval": { + "lower": -0.032, + "upper": 0.032, + "method": "bootstrap" + } + } }, "generation_config": { "additional_details": { @@ -129,21 +153,28 @@ } }, { - "evaluation_name": "Management Consulting Pass@1", + "evaluation_name": "Overall Pass@8", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Management consulting world Pass@1.", + "evaluation_description": "Overall Pass@8 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.227 + "score": 0.4, + "uncertainty": { + "confidence_interval": { + "lower": -0.044, + "upper": 0.044, + "method": "bootstrap" + } + } }, "generation_config": { "additional_details": { @@ -152,21 +183,21 @@ } }, { - "evaluation_name": "Corporate Law Pass@1", + "evaluation_name": "Overall Mean Score", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Corporate law world Pass@1.", + "evaluation_description": "Overall mean rubric score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.189 + "score": 0.387 }, "generation_config": { "additional_details": { @@ -175,75 +206,44 @@ } }, { - "evaluation_name": "Corporate Lawyer Mean Score", + "evaluation_name": "Investment Banking Pass@1", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Corporate lawyer world mean score.", + "evaluation_description": "Investment banking world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.443 + "score": 0.273 }, "generation_config": { "additional_details": { "run_setting": "High" } } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - }, - { - "evaluation_id": "ace/openai_gpt-5.2/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ + }, { - "evaluation_name": "Overall Score", + "evaluation_name": "Management Consulting Pass@1", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Overall ACE score across all consumer-task domains.", + "evaluation_description": "Management consulting world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.515, - "uncertainty": { - "confidence_interval": { - "lower": -0.032, - "upper": 0.032, - "method": "bootstrap" - } - } + "score": 0.227 }, "generation_config": { "additional_details": { @@ -252,21 +252,21 @@ } }, { - "evaluation_name": "Food Score", + "evaluation_name": "Corporate Law Pass@1", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Food domain score.", + "evaluation_description": "Corporate law world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.65 + "score": 0.189 }, "generation_config": { "additional_details": { @@ -275,21 +275,21 @@ } }, { - "evaluation_name": "Gaming Score", + "evaluation_name": "Corporate Lawyer Mean Score", "source_data": { - "dataset_name": "ace", + "dataset_name": "apex-agents", "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" + "hf_repo": "mercor/apex-agents" }, "metric_config": { - "evaluation_description": "Gaming domain score.", + "evaluation_description": "Corporate lawyer world mean score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1 }, "score_details": { - "score": 0.578 + "score": 0.443 }, "generation_config": { "additional_details": { diff --git a/data/models/openassistant_reward-model-deberta-v3-large-v2.json b/data/models/openassistant_reward-model-deberta-v3-large-v2.json index b28ca5c5700af0f5cf22d77dcb1c4fea033ecc2d..cf1ba02f6dd573bf3e0770ff614660b73e92dbb8 100644 --- a/data/models/openassistant_reward-model-deberta-v3-large-v2.json +++ b/data/models/openassistant_reward-model-deberta-v3-large-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6126 + "score": 0.32 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8939 + "score": 0.3853 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4518 + "score": 0.2687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5027 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7338 + "score": 0.3667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3855 + "score": 0.2768 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5836 + "score": 0.12 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.32 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3853 + "score": 0.6126 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2687 + "score": 0.8939 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.4518 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3667 + "score": 0.7338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.3855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.12 + "score": 0.5836 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openbmb_eurus-rm-7b.json b/data/models/openbmb_eurus-rm-7b.json index 44637ca17276f082725f61866dcfa3229cc54274..e1154a660c89f219cec7ec844602d9d8fdfa08ad 100644 --- a/data/models/openbmb_eurus-rm-7b.json +++ b/data/models/openbmb_eurus-rm-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8159 + "score": 0.5806 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9804 + "score": 0.6 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5683 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8135 + "score": 0.6267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8633 + "score": 0.7475 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7172 + "score": 0.5972 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5806 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6 + "score": 0.8159 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9804 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5683 + "score": 0.6557 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6267 + "score": 0.8135 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7475 + "score": 0.8633 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5972 + "score": 0.7172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v1.0-cost.json b/data/models/pku-alignment_beaver-7b-v1.0-cost.json index 8e786484059e3101c1249c1cce8c4b2be81faa5a..3777eba3edfdc470c669a503ac85994bf8139135 100644 --- a/data/models/pku-alignment_beaver-7b-v1.0-cost.json +++ b/data/models/pku-alignment_beaver-7b-v1.0-cost.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5798 + "score": 0.3332 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6173 + "score": 0.3263 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.2313 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3989 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7351 + "score": 0.7589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5482 + "score": 0.2939 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3332 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.5798 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2313 + "score": 0.6173 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3989 + "score": 0.4232 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7589 + "score": 0.7351 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2939 + "score": 0.5482 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.57 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/prithivmlmods_calcium-opus-14b-elite.json b/data/models/prithivmlmods_calcium-opus-14b-elite.json index 746fd41957e7795e0f0f3751181ce9bfef3d8e70..89a0dd9277acb0b40655f39809af8ab4d96076ef 100644 --- a/data/models/prithivmlmods_calcium-opus-14b-elite.json +++ b/data/models/prithivmlmods_calcium-opus-14b-elite.json @@ -5,7 +5,7 @@ "developer": "prithivMLmods", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.766" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6064 + "score": 0.6052 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6296 + "score": 0.6317 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3708 + "score": 0.4789 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3742 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4873 + "score": 0.486 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5307 + "score": 0.5302 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6052 + "score": 0.6064 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6317 + "score": 0.6296 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.3708 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3742 + "score": 0.3733 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.486 + "score": 0.4873 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5302 + "score": 0.5307 } } ], diff --git a/data/models/qingy2019_qwen2.5-math-14b-instruct.json b/data/models/qingy2019_qwen2.5-math-14b-instruct.json index d07b0024c79975f453423bba6d277d902ff3056c..21a12461d0cce8974037575db54605a0da19b661 100644 --- a/data/models/qingy2019_qwen2.5-math-14b-instruct.json +++ b/data/models/qingy2019_qwen2.5-math-14b-instruct.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.6005 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.635 + "score": 0.6356 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.2764 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3725 + "score": 0.3691 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5331 + "score": 0.5339 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6005 + "score": 0.6066 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6356 + "score": 0.635 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2764 + "score": 0.3716 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3691 + "score": 0.3725 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5339 + "score": 0.5331 } } ], diff --git a/data/models/ray2333_grm-llama3-8b-distill.json b/data/models/ray2333_grm-llama3-8b-distill.json index 28bee941f1d39ce464ca73210ca8cbef67a5d082..a698e2dc0120e472c95dbd375c2ce72c243acf20 100644 --- a/data/models/ray2333_grm-llama3-8b-distill.json +++ b/data/models/ray2333_grm-llama3-8b-distill.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.589 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5874 + "score": 0.8464 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5902 + "score": 0.6842 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7222 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6727 + "score": 0.9133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5743 + "score": 0.7209 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8464 + "score": 0.589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5874 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6842 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5902 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9133 + "score": 0.6727 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7209 + "score": 0.5743 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json b/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json index 569e87e2d1650530a990e95b347a974efc9f47d2..294d09da30917c4eff496ef892b03ff9d96db067 100644 --- a/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json +++ b/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6766 + "score": 0.9154 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6274 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.8618 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5847 + "score": 0.9081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.9362 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8929 + "score": 0.6766 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6824 + "score": 0.6274 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9154 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.5847 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8618 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9081 + "score": 0.8929 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9362 + "score": 0.6824 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/ray2333_grm-llama3-8b-sftreg.json b/data/models/ray2333_grm-llama3-8b-sftreg.json index 15f35c842dfd5169bfd85bae53f76a90d5850421..bd70639489991e3e8880cad041284119bd35ecc9 100644 --- a/data/models/ray2333_grm-llama3-8b-sftreg.json +++ b/data/models/ray2333_grm-llama3-8b-sftreg.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6089 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6189 + "score": 0.8542 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5792 + "score": 0.6776 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7867 + "score": 0.8919 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6828 + "score": 0.9229 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5981 + "score": 0.7309 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8542 + "score": 0.6089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.986 + "score": 0.6189 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5792 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8919 + "score": 0.7867 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9229 + "score": 0.6828 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7309 + "score": 0.5981 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/recoilme_gemma-2-ataraxy-gemmasutra-9b-slerp.json b/data/models/recoilme_gemma-2-ataraxy-gemmasutra-9b-slerp.json index f26fe1b9038e8edc381d5411187a5c6b2c7cbf6e..07191bcaa576a8b8b67650d88a341b91a6316705 100644 --- a/data/models/recoilme_gemma-2-ataraxy-gemmasutra-9b-slerp.json +++ b/data/models/recoilme_gemma-2-ataraxy-gemmasutra-9b-slerp.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7649 + "score": 0.2854 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5974 + "score": 0.5984 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0174 + "score": 0.1005 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3305 + "score": 0.3297 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4245 + "score": 0.4607 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4207 + "score": 0.4162 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2854 + "score": 0.7649 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5984 + "score": 0.5974 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1005 + "score": 0.0174 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3297 + "score": 0.3305 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4607 + "score": 0.4245 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.4207 } } ], diff --git a/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json b/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json index 83bfdb05cda442c3bac351d424fa26cf0ae864fe..34121094195afd5811c50d769ba5e5e07669b30d 100644 --- a/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json +++ b/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7592 + "score": 0.2747 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6026 + "score": 0.6031 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0529 + "score": 0.0831 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.3305 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.4686 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4163 + "score": 0.4122 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2747 + "score": 0.7592 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.6026 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0831 + "score": 0.0529 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3305 + "score": 0.3289 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4686 + "score": 0.4099 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4122 + "score": 0.4163 } } ], diff --git a/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json b/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json index d5cadb462419994ffdabe8b6cdbd73be57e99fa8..812871dcfb75c54e3765d2c8d03eeadc60d898aa 100644 --- a/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json +++ b/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5761 + "score": 0.7439 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.602 + "score": 0.5993 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1888 + "score": 0.0876 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3372 + "score": 0.3238 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4632 + "score": 0.4204 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4039 + "score": 0.4072 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7439 + "score": 0.5761 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5993 + "score": 0.602 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0876 + "score": 0.1888 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3372 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4204 + "score": 0.4632 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4072 + "score": 0.4039 } } ], diff --git a/data/models/replete-ai_replete-llm-qwen2-7b.json b/data/models/replete-ai_replete-llm-qwen2-7b.json index 627d67572ee0cd0135e173767115d5ee7360b5e6..e51b10d1bb4ff7d76603042a5cdf25b16b9bc87f 100644 --- a/data/models/replete-ai_replete-llm-qwen2-7b.json +++ b/data/models/replete-ai_replete-llm-qwen2-7b.json @@ -5,7 +5,7 @@ "developer": "Replete-AI", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0905 + "score": 0.0932 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2985 + "score": 0.2977 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2475 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3848 + "score": 0.3941 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1158 + "score": 0.1157 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0932 + "score": 0.0905 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2985 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2475 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3941 + "score": 0.3848 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1157 + "score": 0.1158 } } ], diff --git a/data/models/riaz_finellama-3.1-8b.json b/data/models/riaz_finellama-3.1-8b.json index 926e4908b22e1e63c1ff9b2a4076f85fbbd8288e..554fa555a5c7288aadab0b51655132daffa7eed6 100644 --- a/data/models/riaz_finellama-3.1-8b.json +++ b/data/models/riaz_finellama-3.1-8b.json @@ -5,7 +5,7 @@ "developer": "riaz", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4137 + "score": 0.4373 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4565 + "score": 0.4586 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0453 + "score": 0.0514 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2752 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3776 + "score": 0.3763 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2978 + "score": 0.2964 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4373 + "score": 0.4137 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4586 + "score": 0.4565 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0514 + "score": 0.0453 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.276 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3763 + "score": 0.3776 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2964 + "score": 0.2978 } } ], diff --git a/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json b/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json index 9d341a3c927e4d25f7836c99807ee069a16cbedc..12ab734b7e3e0cffe5b67bb178df3f51d09ec598 100644 --- a/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json +++ b/data/models/rombodawg_rombos-llm-v2.5.1-qwen-3b.json @@ -5,7 +5,7 @@ "developer": "rombodawg", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.397" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2595 + "score": 0.2566 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3884 + "score": 0.39 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0914 + "score": 0.1208 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.2626 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2719 + "score": 0.2741 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2566 + "score": 0.2595 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.39 + "score": 0.3884 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1208 + "score": 0.0914 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2626 + "score": 0.2743 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2741 + "score": 0.2719 } } ], diff --git a/data/models/sao10k_l3-70b-euryale-v2.1.json b/data/models/sao10k_l3-70b-euryale-v2.1.json index c24d09a959df8d13e2602095ebcf1b7a3f8d330c..66fd342cda2fcd43939fb102adcaae988f1eb857 100644 --- a/data/models/sao10k_l3-70b-euryale-v2.1.json +++ b/data/models/sao10k_l3-70b-euryale-v2.1.json @@ -5,7 +5,7 @@ "developer": "Sao10K", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "70.554" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7384 + "score": 0.7281 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6471 + "score": 0.6503 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2137 + "score": 0.2243 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4209 + "score": 0.4196 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5104 + "score": 0.5096 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7281 + "score": 0.7384 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6503 + "score": 0.6471 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2243 + "score": 0.2137 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4196 + "score": 0.4209 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5096 + "score": 0.5104 } } ], diff --git a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json index 988f599afd76311b87e49f16160d0ca534436cdd..4a85b178092e7e5acaf395eb054cb8adc61e2891 100644 --- a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json +++ b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7249 + "score": 0.9499 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7558 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6448 + "score": 0.9378 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.9903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9131 + "score": 0.7249 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7633 + "score": 0.7558 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9499 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.6448 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9079 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9378 + "score": 0.9131 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9903 + "score": 0.7633 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json b/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json index 52952d2b07ce3bbc9c007749585e45b758ae4213..4109d07feac30d842a656539e17ae764cf441c79 100644 --- a/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json +++ b/data/models/skywork_skywork-reward-gemma-2-27b-v0.2.json @@ -142,10 +142,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -164,128 +164,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9426 + "score": 0.7531 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9609 + "score": 0.7674 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8991 + "score": 0.375 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9297 + "score": 0.6721 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9807 + "score": 0.9689 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7531 + "score": 0.9172 }, "source_data": { "dataset_name": "RewardBench 2", @@ -294,111 +270,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7674 + "score": 0.8182 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B-v0.2/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.9426 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6721 + "score": 0.9609 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9689 + "score": 0.8991 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9172 + "score": 0.9297 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8182 + "score": 0.9807 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json index 7fb384620bce09b34b0d31ff763d2b1e971d0558..5d50cd0f56f0ac33470efc8c81f51c02a702598e 100644 --- a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json +++ b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json @@ -5,7 +5,7 @@ "developer": "tanliboy", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Gemma2ForCausalLM", "params_billions": "9.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1829 + "score": 0.4501 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5488 + "score": 0.5472 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0944 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.3138 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4056 + "score": 0.4017 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3792 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4501 + "score": 0.1829 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5472 + "score": 0.5488 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0944 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3104 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4017 + "score": 0.4056 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3792 + "score": 0.3805 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json index 0736460b872bea97a209be68cef1f113fa7d9f3d..f3e37b204fa779a9e21a0521a813b464c3fe641b 100644 --- a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json +++ b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2678 + "score": 0.6496 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4429 + "score": 0.4774 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0566 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.302 + "score": 0.3104 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3909 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2927 + "score": 0.3382 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6496 + "score": 0.2678 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4774 + "score": 0.4429 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0566 + "score": 0.0521 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.302 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3909 + "score": 0.3959 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3382 + "score": 0.2927 } } ], diff --git a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json index f9d2b14f0a0f79f11e39957c0f38b89aa1e78ac9..564c2ceeb0768d947ec7e8507c351558f76e3907 100644 --- a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json +++ b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.2498 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.3642 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.275 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3497 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.24 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3281 + "score": 0.2384 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6564 + "score": 0.0315 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2498 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3642 + "score": 0.5027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3497 + "score": 0.3728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.24 + "score": 0.4149 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2384 + "score": 0.3281 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0315 + "score": 0.6564 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/weqweasdas_rm-gemma-7b.json b/data/models/weqweasdas_rm-gemma-7b.json index 9c3f18f2e5d3851df32fe75b7866a4a718dea91c..6b6f7087683616d43cac5b096d07a117476b2184 100644 --- a/data/models/weqweasdas_rm-gemma-7b.json +++ b/data/models/weqweasdas_rm-gemma-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-7B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6967 + "score": 0.4826 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.4926 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4978 + "score": 0.3937 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6066 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5784 + "score": 0.4822 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7362 + "score": 0.497 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7069 + "score": 0.4232 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-7B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4826 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4926 + "score": 0.6967 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3937 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.4978 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4822 + "score": 0.5784 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.497 + "score": 0.7362 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.7069 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/xai_grok-3-mini.json b/data/models/xai_grok-3-mini.json index 7fda2d0643a4e1f5eb9d98f1dc44e9ec85970d5d..6f7e913322e4d12afd1b4e9815b3c829b5eb051d 100644 --- a/data/models/xai_grok-3-mini.json +++ b/data/models/xai_grok-3-mini.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/xai_grok-4.json b/data/models/xai_grok-4.json index 1a7ae0a5e37faeab9841b8816b2f9d9332f7d639..2396de921bef3d760583d06f283159a7a4b7cc87 100644 --- a/data/models/xai_grok-4.json +++ b/data/models/xai_grok-4.json @@ -4,13 +4,13 @@ "id": "xai/grok-4", "developer": "xAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 23.1, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.1, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/xai_grok-code-fast-1.json b/data/models/xai_grok-code-fast-1.json index 80a497dcbe0e067252e23ff5d2567dcecf97e827..3dead5de67b9b087729b3bc778973bf9f0bdd596 100644 --- a/data/models/xai_grok-code-fast-1.json +++ b/data/models/xai_grok-code-fast-1.json @@ -4,13 +4,13 @@ "id": "xai/grok-code-fast-1", "developer": "xAI", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.8, + "score": 14.2, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 14.2, + "score": 25.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/yam-peleg_hebrew-mistral-7b-200k.json b/data/models/yam-peleg_hebrew-mistral-7b-200k.json index 1a70a6eb9b375af7d3ae622e407043222ccb78a0..baae674a05af479a549c721b22d1c0240e173193 100644 --- a/data/models/yam-peleg_hebrew-mistral-7b-200k.json +++ b/data/models/yam-peleg_hebrew-mistral-7b-200k.json @@ -5,7 +5,7 @@ "developer": "yam-peleg", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "7.504" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.177 + "score": 0.1856 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3411 + "score": 0.4149 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.0234 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.276 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.374 + "score": 0.3765 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2529 + "score": 0.2573 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1856 + "score": 0.177 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.3411 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0234 + "score": 0.031 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3765 + "score": 0.374 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2573 + "score": 0.2529 } } ],