diff --git a/data/benchmarks/appworld_test_normal.json b/data/benchmarks/appworld_test_normal.json index 4b1f2e226ce0f177163e0b4f9252a3d5461ce07f..c94d49539b453c8b9bbc41ff7e62e5792c6b9b48 100644 --- a/data/benchmarks/appworld_test_normal.json +++ b/data/benchmarks/appworld_test_normal.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "appworld/test_normal": 0.66 + "appworld/test_normal": 0.68 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "appworld/test_normal": 0.071 + "appworld/test_normal": 0.0 } } ] diff --git a/data/benchmarks/browsecompplus.json b/data/benchmarks/browsecompplus.json index 613b29235acc848a4b6fbc21b6e89e5e195bd256..74d10ea2ba0e91972d3d0e7424a2805813842fd2 100644 --- a/data/benchmarks/browsecompplus.json +++ b/data/benchmarks/browsecompplus.json @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "browsecompplus": 0.46 + "browsecompplus": 0.43 } } ] diff --git a/data/benchmarks/hfopenllm_v2.json b/data/benchmarks/hfopenllm_v2.json index f63d6b4e97df8ea6a39802305cc7efdb2683daed..ad203e0b72f6caf04084e48c1b921149ab76e134 100644 --- a/data/benchmarks/hfopenllm_v2.json +++ b/data/benchmarks/hfopenllm_v2.json @@ -3047,12 +3047,12 @@ "name": "AetherTOT", "developer": "Daemontatox", "scores": { - "IFEval": 0.4383, - "BBH": 0.5034, - "MATH Level 5": 0.1443, + "IFEval": 0.4398, + "BBH": 0.5066, + "MATH Level 5": 0.1488, "GPQA": 0.3238, - "MUSR": 0.4052, - "MMLU-PRO": 0.3778 + "MUSR": 0.4079, + "MMLU-PRO": 0.3804 } }, { @@ -3125,12 +3125,12 @@ "name": "DocumentCogito", "developer": "Daemontatox", "scores": { - "IFEval": 0.5064, - "BBH": 0.5112, - "MATH Level 5": 0.1631, - "GPQA": 0.3163, - "MUSR": 0.3973, - "MMLU-PRO": 0.3802 + "IFEval": 0.777, + "BBH": 0.5187, + "MATH Level 5": 0.2198, + "GPQA": 0.2936, + "MUSR": 0.3911, + "MMLU-PRO": 0.3738 } }, { @@ -4009,12 +4009,12 @@ "name": "Llama-3.2-1B-SPIN-iter0", "developer": "DavieLion", "scores": { - "IFEval": 0.1507, - "BBH": 0.293, - "MATH Level 5": 0.0, - "GPQA": 0.2534, + "IFEval": 0.1549, + "BBH": 0.2937, + "MATH Level 5": 0.006, + "GPQA": 0.2576, "MUSR": 0.3565, - "MMLU-PRO": 0.1125 + "MMLU-PRO": 0.1128 } }, { @@ -4321,12 +4321,12 @@ "name": "Llama-3.1-8b-ITA", "developer": "DeepMount00", "scores": { - "IFEval": 0.5365, - "BBH": 0.517, - "MATH Level 5": 0.1707, - "GPQA": 0.3062, - "MUSR": 0.4487, - "MMLU-PRO": 0.396 + "IFEval": 0.7917, + "BBH": 0.5109, + "MATH Level 5": 0.1088, + "GPQA": 0.2878, + "MUSR": 0.4136, + "MMLU-PRO": 0.3876 } }, { @@ -7025,12 +7025,12 @@ "name": "Fireball-Meta-Llama-3.1-8B-Instruct-Agent-0.003-128K-code-ds-auto", "developer": "EpistemeAI", "scores": { - "IFEval": 0.7305, - "BBH": 0.4649, - "MATH Level 5": 0.1397, - "GPQA": 0.2659, - "MUSR": 0.3209, - "MMLU-PRO": 0.348 + "IFEval": 0.7207, + "BBH": 0.461, + "MATH Level 5": 0.1314, + "GPQA": 0.2701, + "MUSR": 0.3432, + "MMLU-PRO": 0.3354 } }, { @@ -7675,12 +7675,12 @@ "name": "Herplete-LLM-Llama-3.1-8b", "developer": "Etherll", "scores": { - "IFEval": 0.6106, - "BBH": 0.5347, - "MATH Level 5": 0.1548, - "GPQA": 0.3146, - "MUSR": 0.3991, - "MMLU-PRO": 0.3752 + "IFEval": 0.4672, + "BBH": 0.5013, + "MATH Level 5": 0.0279, + "GPQA": 0.2861, + "MUSR": 0.386, + "MMLU-PRO": 0.3482 } }, { @@ -8455,12 +8455,12 @@ "name": "Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.3417, - "BBH": 0.3292, - "MATH Level 5": 0.0023, - "GPQA": 0.2576, - "MUSR": 0.3249, - "MMLU-PRO": 0.1638 + "IFEval": 0.3472, + "BBH": 0.3268, + "MATH Level 5": 0.0891, + "GPQA": 0.2517, + "MUSR": 0.3262, + "MMLU-PRO": 0.1641 } }, { @@ -8572,12 +8572,12 @@ "name": "josie-7b-v6.0-step2000", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.7628, - "BBH": 0.5098, - "MATH Level 5": 0.0, - "GPQA": 0.2802, - "MUSR": 0.4579, - "MMLU-PRO": 0.4033 + "IFEval": 0.7598, + "BBH": 0.5107, + "MATH Level 5": 0.4237, + "GPQA": 0.2768, + "MUSR": 0.4539, + "MMLU-PRO": 0.4012 } }, { @@ -8728,12 +8728,12 @@ "name": "Gemma-Ko-Merge-PEFT", "developer": "Gunulhona", "scores": { - "IFEval": 0.288, - "BBH": 0.5154, + "IFEval": 0.4441, + "BBH": 0.4863, "MATH Level 5": 0.0, - "GPQA": 0.3247, - "MUSR": 0.408, - "MMLU-PRO": 0.3817 + "GPQA": 0.307, + "MUSR": 0.3986, + "MMLU-PRO": 0.3098 } }, { @@ -9170,12 +9170,12 @@ "name": "SmolLM2-360M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.083, - "BBH": 0.3053, - "MATH Level 5": 0.0083, - "GPQA": 0.2651, - "MUSR": 0.3423, - "MMLU-PRO": 0.1126 + "IFEval": 0.3842, + "BBH": 0.3144, + "MATH Level 5": 0.0151, + "GPQA": 0.255, + "MUSR": 0.3461, + "MMLU-PRO": 0.1117 } }, { @@ -13057,12 +13057,12 @@ "name": "SpydazWeb_AI_HumanAI_012_INSTRUCT_XA", "developer": "LeroyDyer", "scores": { - "IFEval": 0.3579, - "BBH": 0.4477, - "MATH Level 5": 0.0423, - "GPQA": 0.3096, - "MUSR": 0.4134, - "MMLU-PRO": 0.2376 + "IFEval": 0.3798, + "BBH": 0.4483, + "MATH Level 5": 0.04, + "GPQA": 0.3129, + "MUSR": 0.4148, + "MMLU-PRO": 0.2389 } }, { @@ -14305,12 +14305,12 @@ "name": "Llama-3-8B-Magpie-Align-v0.1", "developer": "Magpie-Align", "scores": { - "IFEval": 0.4027, - "BBH": 0.4789, - "MATH Level 5": 0.0461, - "GPQA": 0.2768, - "MUSR": 0.3087, - "MMLU-PRO": 0.3001 + "IFEval": 0.4118, + "BBH": 0.4811, + "MATH Level 5": 0.034, + "GPQA": 0.2752, + "MUSR": 0.3047, + "MMLU-PRO": 0.3006 } }, { @@ -17204,12 +17204,12 @@ "name": "code-yi", "developer": "Omkar1102", "scores": { - "IFEval": 0.2148, - "BBH": 0.276, + "IFEval": 0.2254, + "BBH": 0.275, "MATH Level 5": 0.0, - "GPQA": 0.2508, - "MUSR": 0.3802, - "MMLU-PRO": 0.1126 + "GPQA": 0.2576, + "MUSR": 0.3762, + "MMLU-PRO": 0.1123 } }, { @@ -18141,11 +18141,11 @@ "developer": "PrimeIntellect", "scores": { "IFEval": 0.1757, - "BBH": 0.274, + "BBH": 0.276, "MATH Level 5": 0.0, - "GPQA": 0.25, - "MUSR": 0.3753, - "MMLU-PRO": 0.112 + "GPQA": 0.2534, + "MUSR": 0.3339, + "MMLU-PRO": 0.1123 } }, { @@ -19986,12 +19986,12 @@ "name": "Replete-LLM-Qwen2-7b", "developer": "Replete-AI", "scores": { - "IFEval": 0.0905, - "BBH": 0.2985, + "IFEval": 0.0932, + "BBH": 0.2977, "MATH Level 5": 0.0, - "GPQA": 0.2534, - "MUSR": 0.3848, - "MMLU-PRO": 0.1158 + "GPQA": 0.2475, + "MUSR": 0.3941, + "MMLU-PRO": 0.1157 } }, { @@ -21130,12 +21130,12 @@ "name": "L3-70B-Euryale-v2.1", "developer": "Sao10K", "scores": { - "IFEval": 0.7384, - "BBH": 0.6471, - "MATH Level 5": 0.2137, + "IFEval": 0.7281, + "BBH": 0.6503, + "MATH Level 5": 0.2243, "GPQA": 0.3314, - "MUSR": 0.4209, - "MMLU-PRO": 0.5104 + "MUSR": 0.4196, + "MMLU-PRO": 0.5096 } }, { @@ -25069,12 +25069,12 @@ "name": "Llama3.1-8B-Cobalt", "developer": "ValiantLabs", "scores": { - "IFEval": 0.7168, - "BBH": 0.4911, - "MATH Level 5": 0.1533, - "GPQA": 0.2861, - "MUSR": 0.3512, - "MMLU-PRO": 0.3663 + "IFEval": 0.3496, + "BBH": 0.4947, + "MATH Level 5": 0.1269, + "GPQA": 0.3037, + "MUSR": 0.3959, + "MMLU-PRO": 0.3644 } }, { @@ -25121,12 +25121,12 @@ "name": "Llama3.1-8B-ShiningValiant2", "developer": "ValiantLabs", "scores": { - "IFEval": 0.2678, - "BBH": 0.4429, - "MATH Level 5": 0.0521, - "GPQA": 0.302, - "MUSR": 0.3959, - "MMLU-PRO": 0.2927 + "IFEval": 0.6496, + "BBH": 0.4774, + "MATH Level 5": 0.0566, + "GPQA": 0.3104, + "MUSR": 0.3909, + "MMLU-PRO": 0.3382 } }, { @@ -25654,12 +25654,12 @@ "name": "Qwen2.5-14B-YOYO-1010", "developer": "YOYO-AI", "scores": { - "IFEval": 0.7905, - "BBH": 0.6406, - "MATH Level 5": 0.0, - "GPQA": 0.3163, - "MUSR": 0.4181, - "MMLU-PRO": 0.4944 + "IFEval": 0.5899, + "BBH": 0.654, + "MATH Level 5": 0.4509, + "GPQA": 0.3834, + "MUSR": 0.4744, + "MMLU-PRO": 0.5376 } }, { @@ -26889,12 +26889,12 @@ "name": "Llama-3.1-Storm-8B", "developer": "akjindal53244", "scores": { - "IFEval": 0.8051, - "BBH": 0.5189, - "MATH Level 5": 0.1722, - "GPQA": 0.3263, + "IFEval": 0.8033, + "BBH": 0.5196, + "MATH Level 5": 0.1624, + "GPQA": 0.3096, "MUSR": 0.4028, - "MMLU-PRO": 0.3803 + "MMLU-PRO": 0.3812 } }, { @@ -26954,12 +26954,12 @@ "name": "Llama-3.1-Tulu-3-8B", "developer": "allenai", "scores": { - "IFEval": 0.8255, - "BBH": 0.4061, - "MATH Level 5": 0.2115, - "GPQA": 0.297, + "IFEval": 0.8267, + "BBH": 0.405, + "MATH Level 5": 0.1964, + "GPQA": 0.2987, "MUSR": 0.4175, - "MMLU-PRO": 0.2821 + "MMLU-PRO": 0.2827 } }, { @@ -28449,11 +28449,11 @@ "name": "AMD-Llama-135m", "developer": "amd", "scores": { - "IFEval": 0.1842, - "BBH": 0.2974, - "MATH Level 5": 0.0053, - "GPQA": 0.2525, - "MUSR": 0.378, + "IFEval": 0.1918, + "BBH": 0.2969, + "MATH Level 5": 0.0076, + "GPQA": 0.2584, + "MUSR": 0.3846, "MMLU-PRO": 0.1169 } }, @@ -28709,12 +28709,12 @@ "name": "Arcee-Spark", "developer": "arcee-ai", "scores": { - "IFEval": 0.5718, - "BBH": 0.5481, - "MATH Level 5": 0.114, - "GPQA": 0.3062, - "MUSR": 0.4008, - "MMLU-PRO": 0.3813 + "IFEval": 0.5621, + "BBH": 0.5489, + "MATH Level 5": 0.2953, + "GPQA": 0.307, + "MUSR": 0.4021, + "MMLU-PRO": 0.3822 } }, { @@ -30360,12 +30360,12 @@ "name": "Llama-3.2-3B-Deep-Test", "developer": "bunnycore", "scores": { - "IFEval": 0.1775, - "BBH": 0.295, - "MATH Level 5": 0.0, - "GPQA": 0.2517, - "MUSR": 0.3647, - "MMLU-PRO": 0.1049 + "IFEval": 0.4652, + "BBH": 0.4531, + "MATH Level 5": 0.1284, + "GPQA": 0.2643, + "MUSR": 0.3394, + "MMLU-PRO": 0.3152 } }, { @@ -31647,12 +31647,12 @@ "name": "dolphin-2.9.2-Phi-3-Medium-abliterated", "developer": "cognitivecomputations", "scores": { - "IFEval": 0.4124, - "BBH": 0.6383, - "MATH Level 5": 0.182, - "GPQA": 0.3289, - "MUSR": 0.4349, - "MMLU-PRO": 0.4525 + "IFEval": 0.3613, + "BBH": 0.6123, + "MATH Level 5": 0.1239, + "GPQA": 0.328, + "MUSR": 0.4112, + "MMLU-PRO": 0.4494 } }, { @@ -34663,12 +34663,12 @@ "name": "gemma-2-2b", "developer": "Google", "scores": { - "IFEval": 0.1993, - "BBH": 0.3656, - "MATH Level 5": 0.0287, + "IFEval": 0.2018, + "BBH": 0.3709, + "MATH Level 5": 0.0302, "GPQA": 0.2626, - "MUSR": 0.4232, - "MMLU-PRO": 0.218 + "MUSR": 0.4219, + "MMLU-PRO": 0.2217 } }, { @@ -37705,12 +37705,12 @@ "name": "Kosmos-EVAA-Fusion-8B", "developer": "jaspionjader", "scores": { - "IFEval": 0.4418, - "BBH": 0.5406, - "MATH Level 5": 0.1352, - "GPQA": 0.3062, + "IFEval": 0.4345, + "BBH": 0.5419, + "MATH Level 5": 0.1292, + "GPQA": 0.3087, "MUSR": 0.4277, - "MMLU-PRO": 0.386 + "MMLU-PRO": 0.3854 } }, { @@ -42359,12 +42359,12 @@ "name": "Mistral-v0.3-7B-ORPO", "developer": "llmat", "scores": { - "IFEval": 0.364, - "BBH": 0.4005, - "MATH Level 5": 0.0015, - "GPQA": 0.2693, - "MUSR": 0.3529, - "MMLU-PRO": 0.2301 + "IFEval": 0.377, + "BBH": 0.3978, + "MATH Level 5": 0.0242, + "GPQA": 0.2668, + "MUSR": 0.3555, + "MMLU-PRO": 0.2278 } }, { @@ -43893,12 +43893,12 @@ "name": "Meta-Llama-3-8B-Instruct", "developer": "meta-llama", "scores": { - "IFEval": 0.7408, - "BBH": 0.4989, - "MATH Level 5": 0.0869, - "GPQA": 0.2592, - "MUSR": 0.3568, - "MMLU-PRO": 0.3664 + "IFEval": 0.4782, + "BBH": 0.491, + "MATH Level 5": 0.0914, + "GPQA": 0.2928, + "MUSR": 0.3805, + "MMLU-PRO": 0.3591 } }, { @@ -43984,12 +43984,12 @@ "name": "Phi-3-mini-4k-instruct", "developer": "microsoft", "scores": { - "IFEval": 0.5477, - "BBH": 0.5491, - "MATH Level 5": 0.1639, - "GPQA": 0.3322, - "MUSR": 0.4284, - "MMLU-PRO": 0.4022 + "IFEval": 0.5613, + "BBH": 0.5676, + "MATH Level 5": 0.1163, + "GPQA": 0.3196, + "MUSR": 0.395, + "MMLU-PRO": 0.3866 } }, { @@ -45076,12 +45076,12 @@ "name": "Mistral-Nemo-Kurdish-Instruct", "developer": "nazimali", "scores": { - "IFEval": 0.4964, - "BBH": 0.4699, - "MATH Level 5": 0.0045, - "GPQA": 0.2827, - "MUSR": 0.3979, - "MMLU-PRO": 0.3063 + "IFEval": 0.486, + "BBH": 0.4721, + "MATH Level 5": 0.0846, + "GPQA": 0.2844, + "MUSR": 0.4006, + "MMLU-PRO": 0.3087 } }, { @@ -47611,12 +47611,12 @@ "name": "wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue", "developer": "ontocord", "scores": { - "IFEval": 0.1128, - "BBH": 0.3171, - "MATH Level 5": 0.0113, - "GPQA": 0.2685, - "MUSR": 0.346, - "MMLU-PRO": 0.1129 + "IFEval": 0.1162, + "BBH": 0.3184, + "MATH Level 5": 0.0076, + "GPQA": 0.2634, + "MUSR": 0.3447, + "MMLU-PRO": 0.1124 } }, { @@ -50861,12 +50861,12 @@ "name": "Oracle-14B", "developer": "qingy2019", "scores": { - "IFEval": 0.2358, - "BBH": 0.4612, - "MATH Level 5": 0.0642, - "GPQA": 0.2576, - "MUSR": 0.3717, - "MMLU-PRO": 0.2382 + "IFEval": 0.2401, + "BBH": 0.4622, + "MATH Level 5": 0.0725, + "GPQA": 0.2609, + "MUSR": 0.3703, + "MMLU-PRO": 0.2379 } }, { @@ -50874,12 +50874,12 @@ "name": "Qwen2.5-Math-14B-Instruct", "developer": "qingy2019", "scores": { - "IFEval": 0.6066, - "BBH": 0.635, - "MATH Level 5": 0.3716, - "GPQA": 0.3725, + "IFEval": 0.6005, + "BBH": 0.6356, + "MATH Level 5": 0.2764, + "GPQA": 0.3691, "MUSR": 0.4757, - "MMLU-PRO": 0.5331 + "MMLU-PRO": 0.5339 } }, { @@ -51459,12 +51459,12 @@ "name": "FineLlama-3.1-8B", "developer": "riaz", "scores": { - "IFEval": 0.4137, - "BBH": 0.4565, - "MATH Level 5": 0.0453, - "GPQA": 0.276, - "MUSR": 0.3776, - "MMLU-PRO": 0.2978 + "IFEval": 0.4373, + "BBH": 0.4586, + "MATH Level 5": 0.0514, + "GPQA": 0.2752, + "MUSR": 0.3763, + "MMLU-PRO": 0.2964 } }, { @@ -53201,12 +53201,12 @@ "name": "Qwenvergence-14B-v3-Reason", "developer": "sometimesanotion", "scores": { - "IFEval": 0.5278, - "BBH": 0.6557, - "MATH Level 5": 0.3119, - "GPQA": 0.3842, - "MUSR": 0.4754, - "MMLU-PRO": 0.5396 + "IFEval": 0.5367, + "BBH": 0.6561, + "MATH Level 5": 0.358, + "GPQA": 0.3867, + "MUSR": 0.474, + "MMLU-PRO": 0.5395 } }, { @@ -53500,12 +53500,12 @@ "name": "ChatWaifu_v2.0_22B", "developer": "spow12", "scores": { - "IFEval": 0.6511, - "BBH": 0.5926, - "MATH Level 5": 0.1858, - "GPQA": 0.3247, + "IFEval": 0.6517, + "BBH": 0.5908, + "MATH Level 5": 0.2032, + "GPQA": 0.3238, "MUSR": 0.3842, - "MMLU-PRO": 0.3836 + "MMLU-PRO": 0.3812 } }, { @@ -54345,12 +54345,12 @@ "name": "lambda-gemma-2-9b-dpo", "developer": "tanliboy", "scores": { - "IFEval": 0.1829, - "BBH": 0.5488, - "MATH Level 5": 0.0, - "GPQA": 0.3104, - "MUSR": 0.4056, - "MMLU-PRO": 0.3805 + "IFEval": 0.4501, + "BBH": 0.5472, + "MATH Level 5": 0.0944, + "GPQA": 0.3138, + "MUSR": 0.4017, + "MMLU-PRO": 0.3792 } }, { @@ -56997,12 +56997,12 @@ "name": "BagelMIsteryTour-v2-8x7B", "developer": "ycros", "scores": { - "IFEval": 0.6262, - "BBH": 0.5142, - "MATH Level 5": 0.0937, - "GPQA": 0.3079, - "MUSR": 0.4138, - "MMLU-PRO": 0.3481 + "IFEval": 0.5994, + "BBH": 0.5159, + "MATH Level 5": 0.0785, + "GPQA": 0.3045, + "MUSR": 0.4203, + "MMLU-PRO": 0.3473 } }, { diff --git a/data/benchmarks/livecodebenchpro.json b/data/benchmarks/livecodebenchpro.json index 449e36cb0d36e98776192be2ede4fd274d74ec7e..f8f77204727bc54bca1bbaaa204ca44d2b265f12 100644 --- a/data/benchmarks/livecodebenchpro.json +++ b/data/benchmarks/livecodebenchpro.json @@ -255,9 +255,9 @@ "name": "o4-mini-2025-04-16", "developer": "OpenAI", "scores": { - "Hard Problems": 0.014084507042253521, - "Medium Problems": 0.30985915492957744, - "Easy Problems": 0.8873239436619719 + "Hard Problems": 0.0143, + "Medium Problems": 0.2923, + "Easy Problems": 0.8571 } }, { diff --git a/data/benchmarks/reward-bench.json b/data/benchmarks/reward-bench.json index 083fee6e840e33c85c4fe87b781938b7ad720c00..a29ee6207f83073111bedbdb1ff7d98e527e8df6 100644 --- a/data/benchmarks/reward-bench.json +++ b/data/benchmarks/reward-bench.json @@ -81,17 +81,17 @@ "name": "CIR-AMS/BTRM_Qwen2_7b_0613", "developer": "CIR-AMS", "scores": { - "Score": 0.8172, + "Score": 0.5736, + "Chat": 0.9749, + "Chat Hard": 0.5724, + "Safety": 0.7178, + "Reasoning": 0.8775, + "Prior Sets (0.5 weight)": 0.7029, "Factuality": 0.5347, "Precise IF": 0.3563, "Math": 0.6066, - "Safety": 0.9014, "Focus": 0.5737, - "Ties": 0.6527, - "Chat": 0.9749, - "Chat Hard": 0.5724, - "Reasoning": 0.8775, - "Prior Sets (0.5 weight)": 0.7029 + "Ties": 0.6527 } }, { @@ -609,17 +609,17 @@ "name": "PKU-Alignment/beaver-7b-v1.0-cost", "developer": "PKU-Alignment", "scores": { - "Score": 0.5798, + "Score": 0.3332, + "Chat": 0.6173, + "Chat Hard": 0.4232, + "Safety": 0.7589, + "Reasoning": 0.5482, + "Prior Sets (0.5 weight)": 0.57, "Factuality": 0.3263, "Precise IF": 0.2313, "Math": 0.3989, - "Safety": 0.7351, "Focus": 0.2939, - "Ties": -0.01, - "Chat": 0.6173, - "Chat Hard": 0.4232, - "Reasoning": 0.5482, - "Prior Sets (0.5 weight)": 0.57 + "Ties": -0.01 } }, { @@ -938,17 +938,17 @@ "name": "Ray2333/GRM-llama3-8B-distill", "developer": "Ray2333", "scores": { - "Score": 0.589, - "Chat": 0.9832, - "Chat Hard": 0.6842, - "Safety": 0.7222, - "Reasoning": 0.9133, - "Prior Sets (0.5 weight)": 0.7209, + "Score": 0.8464, "Factuality": 0.5874, "Precise IF": 0.3875, "Math": 0.5902, + "Safety": 0.8676, "Focus": 0.6727, - "Ties": 0.5743 + "Ties": 0.5743, + "Chat": 0.9832, + "Chat Hard": 0.6842, + "Reasoning": 0.9133, + "Prior Sets (0.5 weight)": 0.7209 } }, { @@ -1139,16 +1139,16 @@ "name": "Skywork/Skywork-Reward-Gemma-2-27B", "developer": "Skywork", "scores": { - "Score": 0.938, + "Score": 0.7576, + "Chat": 0.9581, + "Chat Hard": 0.9145, + "Safety": 0.9422, + "Reasoning": 0.9606, "Factuality": 0.7368, "Precise IF": 0.4031, "Math": 0.7049, - "Safety": 0.9189, "Focus": 0.9323, - "Ties": 0.8261, - "Chat": 0.9581, - "Chat Hard": 0.9145, - "Reasoning": 0.9606 + "Ties": 0.8261 } }, { @@ -1173,16 +1173,16 @@ "name": "Skywork/Skywork-Reward-Llama-3.1-8B", "developer": "Skywork", "scores": { - "Score": 0.7314, - "Chat": 0.9581, - "Chat Hard": 0.8728, - "Safety": 0.9333, - "Reasoning": 0.962, + "Score": 0.9252, "Factuality": 0.6989, "Precise IF": 0.425, "Math": 0.6284, + "Safety": 0.9081, "Focus": 0.9616, - "Ties": 0.741 + "Ties": 0.741, + "Chat": 0.9581, + "Chat Hard": 0.8728, + "Reasoning": 0.962 } }, { @@ -1305,16 +1305,16 @@ "name": "Skywork/Skywork-VL-Reward-7B", "developer": "Skywork", "scores": { - "Score": 0.9007, + "Score": 0.6885, + "Chat": 0.8994, + "Chat Hard": 0.875, + "Safety": 0.8911, + "Reasoning": 0.9176, "Factuality": 0.6063, "Precise IF": 0.35, "Math": 0.6339, - "Safety": 0.9108, "Focus": 0.8909, - "Ties": 0.7586, - "Chat": 0.8994, - "Chat Hard": 0.875, - "Reasoning": 0.9176 + "Ties": 0.7586 } }, { @@ -1379,10 +1379,10 @@ "name": "ai2/tulu-2-7b-rm-v0-nectar-binarized-3.8m-check...", "developer": "AI2", "scores": { - "Score": 0.6924, - "Chat": 0.9441, - "Chat Hard": 0.3575, - "Safety": 0.7757 + "Score": 0.6895, + "Chat": 0.9385, + "Chat Hard": 0.3706, + "Safety": 0.7595 } }, { @@ -1477,17 +1477,17 @@ "name": "allenai/Llama-3.1-Tulu-3-70B-SFT-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.722, - "Chat": 0.9693, - "Chat Hard": 0.8268, - "Safety": 0.8689, - "Reasoning": 0.8583, - "Prior Sets (0.5 weight)": 0.0, + "Score": 0.8892, "Factuality": 0.8084, "Precise IF": 0.3688, "Math": 0.6776, + "Safety": 0.9027, "Focus": 0.7778, - "Ties": 0.8308 + "Ties": 0.8308, + "Chat": 0.9693, + "Chat Hard": 0.8268, + "Reasoning": 0.8583, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1495,17 +1495,17 @@ "name": "allenai/Llama-3.1-Tulu-3-8B-DPO-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8431, + "Score": 0.687, + "Chat": 0.9553, + "Chat Hard": 0.761, + "Safety": 0.86, + "Reasoning": 0.7898, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.7516, "Precise IF": 0.3875, "Math": 0.6284, - "Safety": 0.8662, "Focus": 0.8545, - "Ties": 0.6397, - "Chat": 0.9553, - "Chat Hard": 0.761, - "Reasoning": 0.7898, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.6397 } }, { @@ -3486,17 +3486,17 @@ "name": "Claude 3 Haiku 20240307", "developer": "Anthropic", "scores": { - "Score": 0.3711, - "Chat": 0.9274, - "Chat Hard": 0.5197, - "Safety": 0.595, - "Reasoning": 0.706, - "Prior Sets (0.5 weight)": 0.6635, + "Score": 0.7289, "Factuality": 0.4042, "Precise IF": 0.2812, "Math": 0.3552, + "Safety": 0.7953, "Focus": 0.501, - "Ties": 0.0899 + "Ties": 0.0899, + "Chat": 0.9274, + "Chat Hard": 0.5197, + "Reasoning": 0.706, + "Prior Sets (0.5 weight)": 0.6635 } }, { @@ -3504,16 +3504,16 @@ "name": "Claude 3 Opus 20240229", "developer": "Anthropic", "scores": { - "Score": 0.8008, + "Score": 0.5744, + "Chat": 0.9469, + "Chat Hard": 0.6031, + "Safety": 0.8378, + "Reasoning": 0.7868, "Factuality": 0.5389, "Precise IF": 0.3312, "Math": 0.5137, - "Safety": 0.8662, "Focus": 0.6646, - "Ties": 0.5601, - "Chat": 0.9469, - "Chat Hard": 0.6031, - "Reasoning": 0.7868 + "Ties": 0.5601 } }, { @@ -3801,16 +3801,16 @@ "name": "internlm/internlm2-1_8b-reward", "developer": "internlm", "scores": { - "Score": 0.8217, + "Score": 0.3902, + "Chat": 0.9358, + "Chat Hard": 0.6623, + "Safety": 0.4711, + "Reasoning": 0.8724, "Factuality": 0.2758, "Precise IF": 0.3625, "Math": 0.4426, - "Safety": 0.8162, "Focus": 0.596, - "Ties": 0.1934, - "Chat": 0.9358, - "Chat Hard": 0.6623, - "Reasoning": 0.8724 + "Ties": 0.1934 } }, { @@ -4014,16 +4014,16 @@ "name": "nicolinho/QRM-Gemma-2-27B", "developer": "nicolinho", "scores": { - "Score": 0.9444, + "Score": 0.7667, + "Chat": 0.9665, + "Chat Hard": 0.9013, + "Safety": 0.9578, + "Reasoning": 0.9826, "Factuality": 0.7853, "Precise IF": 0.3719, "Math": 0.6995, - "Safety": 0.927, "Focus": 0.9535, - "Ties": 0.8321, - "Chat": 0.9665, - "Chat Hard": 0.9013, - "Reasoning": 0.9826 + "Ties": 0.8321 } }, { @@ -4219,16 +4219,16 @@ "name": "GPT-4o mini 2024-07-18", "developer": "OpenAI", "scores": { - "Score": 0.5796, - "Chat": 0.9497, - "Chat Hard": 0.6075, - "Safety": 0.7667, - "Reasoning": 0.8374, + "Score": 0.8007, "Factuality": 0.4105, "Precise IF": 0.3438, "Math": 0.5191, + "Safety": 0.8081, "Focus": 0.7414, - "Ties": 0.6962 + "Ties": 0.6962, + "Chat": 0.9497, + "Chat Hard": 0.6075, + "Reasoning": 0.8374 } }, { @@ -4249,17 +4249,17 @@ "name": "openbmb/Eurus-RM-7b", "developer": "openbmb", "scores": { - "Score": 0.8159, + "Score": 0.5806, + "Chat": 0.9804, + "Chat Hard": 0.6557, + "Safety": 0.6267, + "Reasoning": 0.8633, + "Prior Sets (0.5 weight)": 0.7172, "Factuality": 0.6, "Precise IF": 0.3438, "Math": 0.5683, - "Safety": 0.8135, "Focus": 0.7475, - "Ties": 0.5972, - "Chat": 0.9804, - "Chat Hard": 0.6557, - "Reasoning": 0.8633, - "Prior Sets (0.5 weight)": 0.7172 + "Ties": 0.5972 } }, { @@ -4280,17 +4280,17 @@ "name": "openbmb/UltraRM-13b", "developer": "openbmb", "scores": { - "Score": 0.4683, - "Chat": 0.9637, - "Chat Hard": 0.5548, - "Safety": 0.5089, - "Reasoning": 0.6244, - "Prior Sets (0.5 weight)": 0.7294, + "Score": 0.6903, "Factuality": 0.5063, "Precise IF": 0.3312, "Math": 0.5519, + "Safety": 0.5986, "Focus": 0.6081, - "Ties": 0.3036 + "Ties": 0.3036, + "Chat": 0.9637, + "Chat Hard": 0.5548, + "Reasoning": 0.6244, + "Prior Sets (0.5 weight)": 0.7294 } }, { @@ -4370,17 +4370,17 @@ "name": "sfairXC/FsfairX-LLaMA3-RM-v0.1", "developer": "sfairXC", "scores": { - "Score": 0.8338, + "Score": 0.6292, + "Chat": 0.9944, + "Chat Hard": 0.6513, + "Safety": 0.7667, + "Reasoning": 0.8644, + "Prior Sets (0.5 weight)": 0.7492, "Factuality": 0.5916, "Precise IF": 0.4188, "Math": 0.6284, - "Safety": 0.8676, "Focus": 0.7051, - "Ties": 0.6647, - "Chat": 0.9944, - "Chat Hard": 0.6513, - "Reasoning": 0.8644, - "Prior Sets (0.5 weight)": 0.7492 + "Ties": 0.6647 } }, { @@ -4492,17 +4492,17 @@ "name": "weqweasdas/RM-Gemma-2B", "developer": "weqweasdas", "scores": { - "Score": 0.3057, - "Chat": 0.9441, - "Chat Hard": 0.4079, - "Safety": 0.3311, - "Reasoning": 0.7637, - "Prior Sets (0.5 weight)": 0.6652, + "Score": 0.6549, "Factuality": 0.3705, "Precise IF": 0.2812, "Math": 0.4317, + "Safety": 0.4986, "Focus": 0.2343, - "Ties": 0.1851 + "Ties": 0.1851, + "Chat": 0.9441, + "Chat Hard": 0.4079, + "Reasoning": 0.7637, + "Prior Sets (0.5 weight)": 0.6652 } }, { @@ -4541,17 +4541,17 @@ "name": "weqweasdas/RM-Mistral-7B", "developer": "weqweasdas", "scores": { - "Score": 0.596, - "Chat": 0.9665, - "Chat Hard": 0.6053, - "Safety": 0.6911, - "Reasoning": 0.7736, - "Prior Sets (0.5 weight)": 0.753, + "Score": 0.7982, "Factuality": 0.5937, "Precise IF": 0.3438, "Math": 0.5956, + "Safety": 0.8703, "Focus": 0.7293, - "Ties": 0.6226 + "Ties": 0.6226, + "Chat": 0.9665, + "Chat Hard": 0.6053, + "Reasoning": 0.7736, + "Prior Sets (0.5 weight)": 0.753 } }, { @@ -4559,17 +4559,17 @@ "name": "weqweasdas/hh_rlhf_rm_open_llama_3b", "developer": "weqweasdas", "scores": { - "Score": 0.5027, + "Score": 0.2498, + "Chat": 0.8184, + "Chat Hard": 0.3728, + "Safety": 0.24, + "Reasoning": 0.3281, + "Prior Sets (0.5 weight)": 0.6564, "Factuality": 0.3642, "Precise IF": 0.275, "Math": 0.3497, - "Safety": 0.4149, "Focus": 0.2384, - "Ties": 0.0315, - "Chat": 0.8184, - "Chat Hard": 0.3728, - "Reasoning": 0.3281, - "Prior Sets (0.5 weight)": 0.6564 + "Ties": 0.0315 } } ] diff --git a/data/benchmarks/swe-bench.json b/data/benchmarks/swe-bench.json index 3b6c1cd3a38d41c01ec61c5d181997bdb0b7b011..093176be195c26e24230b48d3e9b488139a6d6bf 100644 --- a/data/benchmarks/swe-bench.json +++ b/data/benchmarks/swe-bench.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "swe-bench": 0.65 + "swe-bench": 0.8072 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "swe-bench": 0.57 + "swe-bench": 0.5455 } } ] diff --git a/data/benchmarks/tau-bench-2_airline.json b/data/benchmarks/tau-bench-2_airline.json index 3829696a07bc037642c69851a891e0aeb0e5febf..ebf0dad8ac51717ac2f9f1d57eb468b1c5031207 100644 --- a/data/benchmarks/tau-bench-2_airline.json +++ b/data/benchmarks/tau-bench-2_airline.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/airline": 0.72 + "tau-bench-2/airline": 0.66 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/airline": 0.68 + "tau-bench-2/airline": 0.62 } }, { diff --git a/data/benchmarks/tau-bench-2_retail.json b/data/benchmarks/tau-bench-2_retail.json index 43bf4e72d939794b39171defb2c6b721324ab767..20515090a1536993197150db4c0d92e2f6d83d0d 100644 --- a/data/benchmarks/tau-bench-2_retail.json +++ b/data/benchmarks/tau-bench-2_retail.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/retail": 0.85 + "tau-bench-2/retail": 0.83 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/retail": 0.82 + "tau-bench-2/retail": 0.73 } }, { diff --git a/data/benchmarks/tau-bench-2_telecom.json b/data/benchmarks/tau-bench-2_telecom.json index 05002e54c35c11c6e2530af6d100378b4038ab6c..0371242e1ef28be51ed37461c45171b5f6b938db 100644 --- a/data/benchmarks/tau-bench-2_telecom.json +++ b/data/benchmarks/tau-bench-2_telecom.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/telecom": 0.73 + "tau-bench-2/telecom": 0.6852 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/telecom": 0.55 + "tau-bench-2/telecom": 0.5354 } } ] diff --git a/data/benchmarks/terminal-bench-2.0.json b/data/benchmarks/terminal-bench-2.0.json index 3f1515b599ae57812585a3e55eaf803056a58325..ee5c9bf500efbdc64465ca661d8781cc2133425b 100644 --- a/data/benchmarks/terminal-bench-2.0.json +++ b/data/benchmarks/terminal-bench-2.0.json @@ -5,7 +5,7 @@ "name": "Qwen 3 Coder 480B", "developer": "Alibaba", "scores": { - "terminal-bench-2.0": 23.9 + "terminal-bench-2.0": 27.2 } }, { @@ -13,7 +13,7 @@ "name": "Claude Haiku 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 13.9 + "terminal-bench-2.0": 28.3 } }, { @@ -21,7 +21,7 @@ "name": "Claude Opus 4.1", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 36.9 + "terminal-bench-2.0": 34.8 } }, { @@ -37,7 +37,7 @@ "name": "Claude Opus 4.6", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 74.7 + "terminal-bench-2.0": 62.9 } }, { @@ -61,7 +61,7 @@ "name": "Gemini 2.5 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 15.4 + "terminal-bench-2.0": 17.1 } }, { @@ -77,7 +77,7 @@ "name": "Gemini 3 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 51.0 + "terminal-bench-2.0": 64.3 } }, { @@ -85,7 +85,7 @@ "name": "Gemini 3 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 62.2 + "terminal-bench-2.0": 61.8 } }, { @@ -93,7 +93,7 @@ "name": "Gemini 3.1 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 74.8 + "terminal-bench-2.0": 78.4 } }, { @@ -109,7 +109,7 @@ "name": "MiniMax M2.1", "developer": "MiniMax", "scores": { - "terminal-bench-2.0": 29.2 + "terminal-bench-2.0": 36.6 } }, { @@ -125,7 +125,7 @@ "name": "Kimi K2 Instruct", "developer": "Moonshot AI", "scores": { - "terminal-bench-2.0": 27.8 + "terminal-bench-2.0": 26.7 } }, { @@ -157,7 +157,7 @@ "name": "GPT-5", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 49.6 + "terminal-bench-2.0": 35.2 } }, { @@ -165,7 +165,7 @@ "name": "GPT-5-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 43.4 + "terminal-bench-2.0": 44.3 } }, { @@ -173,7 +173,7 @@ "name": "GPT-5-Mini", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 34.8 + "terminal-bench-2.0": 24.0 } }, { @@ -181,7 +181,7 @@ "name": "GPT-5-Nano", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 7.0 + "terminal-bench-2.0": 7.9 } }, { @@ -221,7 +221,7 @@ "name": "GPT-5.2", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 60.7 + "terminal-bench-2.0": 54.0 } }, { @@ -269,7 +269,7 @@ "name": "Grok Code Fast 1", "developer": "xAI", "scores": { - "terminal-bench-2.0": 25.8 + "terminal-bench-2.0": 14.2 } }, { diff --git a/data/developers/ai2.json b/data/developers/ai2.json index 6ae2e91a5d501e1d313c59819f3bee806d5615b0..498b6154facf655073d48826dd02116d29e34e45 100644 --- a/data/developers/ai2.json +++ b/data/developers/ai2.json @@ -43,10 +43,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6924, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.3575, - "reward-bench/Safety": 0.7757 + "reward-bench/Score": 0.6895, + "reward-bench/Chat": 0.9385, + "reward-bench/Chat Hard": 0.3706, + "reward-bench/Safety": 0.7595 } }, { diff --git a/data/developers/akjindal53244.json b/data/developers/akjindal53244.json index 237ea0357d953fdc2d416f7c27241c406836e723..86acdd918ca996450aa81c0111cbad62da1b8b17 100644 --- a/data/developers/akjindal53244.json +++ b/data/developers/akjindal53244.json @@ -7,12 +7,12 @@ "developer": "akjindal53244", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8051, - "hfopenllm_v2/BBH": 0.5189, - "hfopenllm_v2/MATH Level 5": 0.1722, - "hfopenllm_v2/GPQA": 0.3263, + "hfopenllm_v2/IFEval": 0.8033, + "hfopenllm_v2/BBH": 0.5196, + "hfopenllm_v2/MATH Level 5": 0.1624, + "hfopenllm_v2/GPQA": 0.3096, "hfopenllm_v2/MUSR": 0.4028, - "hfopenllm_v2/MMLU-PRO": 0.3803 + "hfopenllm_v2/MMLU-PRO": 0.3812 } } ] diff --git a/data/developers/alibaba.json b/data/developers/alibaba.json index 759f3a2b29b65da87e85bd7df67845381cdc3145..844401e341e9e7b7e568fb2e5cdb8d2481e4039c 100644 --- a/data/developers/alibaba.json +++ b/data/developers/alibaba.json @@ -7,7 +7,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.9 + "terminal-bench-2.0/terminal-bench-2.0": 27.2 } }, { diff --git a/data/developers/allenai.json b/data/developers/allenai.json index 733bf4fc5c0900a8884a7abb1f0b0f474d01c62a..335d05b967befc327e311cd92223e23391ec53c6 100644 --- a/data/developers/allenai.json +++ b/data/developers/allenai.json @@ -162,17 +162,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.722, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Safety": 0.8689, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8892, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, + "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308 + "reward-bench/Ties": 0.8308, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -181,12 +181,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { @@ -209,17 +209,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8431, + "reward-bench/Score": 0.687, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.86, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6397 } }, { diff --git a/data/developers/amd.json b/data/developers/amd.json index ec85a4364a761e30f1999f359d0d247d8857e139..f58c5a6931b8f7bef3756ba18944919a5d99e792 100644 --- a/data/developers/amd.json +++ b/data/developers/amd.json @@ -7,11 +7,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1842, - "hfopenllm_v2/BBH": 0.2974, - "hfopenllm_v2/MATH Level 5": 0.0053, - "hfopenllm_v2/GPQA": 0.2525, - "hfopenllm_v2/MUSR": 0.378, + "hfopenllm_v2/IFEval": 0.1918, + "hfopenllm_v2/BBH": 0.2969, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/MUSR": 0.3846, "hfopenllm_v2/MMLU-PRO": 0.1169 } } diff --git a/data/developers/anthropic.json b/data/developers/anthropic.json index 21c2bc0153b15a2859b6559b8628567638664450..8e4dd896b5ad3de59cdc15abc7af29c17ad664c2 100644 --- a/data/developers/anthropic.json +++ b/data/developers/anthropic.json @@ -371,17 +371,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.3711, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Safety": 0.595, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635, + "reward-bench/Score": 0.7289, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, + "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899 + "reward-bench/Ties": 0.0899, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635 } }, { @@ -436,16 +436,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.8008, + "reward-bench/Score": 0.5744, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.8378, + "reward-bench/Reasoning": 0.7868, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Reasoning": 0.7868 + "reward-bench/Ties": 0.5601 } }, { @@ -525,7 +525,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 13.9 + "terminal-bench-2.0/terminal-bench-2.0": 28.3 } }, { @@ -650,11 +650,11 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.66, + "appworld_test_normal/appworld/test_normal": 0.68, "browsecompplus/browsecompplus": 0.49, - "swe-bench/swe-bench": 0.65, - "tau-bench-2_airline/tau-bench-2/airline": 0.72, - "tau-bench-2_retail/tau-bench-2/retail": 0.85, + "swe-bench/swe-bench": 0.8072, + "tau-bench-2_airline/tau-bench-2/airline": 0.66, + "tau-bench-2_retail/tau-bench-2/retail": 0.83, "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } }, @@ -664,7 +664,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -682,7 +682,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.7 + "terminal-bench-2.0/terminal-bench-2.0": 62.9 } }, { diff --git a/data/developers/arcee-ai.json b/data/developers/arcee-ai.json index a67ed73a3aff7f3d10c24e6ce76a817fd626a3c8..fb3c26af6dfe804ff7a40fb7fe0b6065ba67af15 100644 --- a/data/developers/arcee-ai.json +++ b/data/developers/arcee-ai.json @@ -49,12 +49,12 @@ "developer": "arcee-ai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5718, - "hfopenllm_v2/BBH": 0.5481, - "hfopenllm_v2/MATH Level 5": 0.114, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4008, - "hfopenllm_v2/MMLU-PRO": 0.3813 + "hfopenllm_v2/IFEval": 0.5621, + "hfopenllm_v2/BBH": 0.5489, + "hfopenllm_v2/MATH Level 5": 0.2953, + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.4021, + "hfopenllm_v2/MMLU-PRO": 0.3822 } }, { diff --git a/data/developers/bunnycore.json b/data/developers/bunnycore.json index 153064e2dea6c3626cb6deb17e2d3191b25ec406..069b8a9f0214975725e28f414b82b07e73072c0f 100644 --- a/data/developers/bunnycore.json +++ b/data/developers/bunnycore.json @@ -287,12 +287,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1775, - "hfopenllm_v2/BBH": 0.295, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1049 + "hfopenllm_v2/IFEval": 0.4652, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.1284, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3394, + "hfopenllm_v2/MMLU-PRO": 0.3152 } }, { diff --git a/data/developers/cir-ams.json b/data/developers/cir-ams.json index df9dcecb6f8fae5d901c2496b58813341797427e..09d0cc390a2584b96dbfe9a5acec25171fe175fa 100644 --- a/data/developers/cir-ams.json +++ b/data/developers/cir-ams.json @@ -7,17 +7,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } } ] diff --git a/data/developers/cognitivecomputations.json b/data/developers/cognitivecomputations.json index 27ef3ede420acfeab83ed5bb754062e32374c41a..292d5e513d244c8ebb441a809017ff16919c9354 100644 --- a/data/developers/cognitivecomputations.json +++ b/data/developers/cognitivecomputations.json @@ -77,12 +77,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4124, - "hfopenllm_v2/BBH": 0.6383, - "hfopenllm_v2/MATH Level 5": 0.182, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4349, - "hfopenllm_v2/MMLU-PRO": 0.4525 + "hfopenllm_v2/IFEval": 0.3613, + "hfopenllm_v2/BBH": 0.6123, + "hfopenllm_v2/MATH Level 5": 0.1239, + "hfopenllm_v2/GPQA": 0.328, + "hfopenllm_v2/MUSR": 0.4112, + "hfopenllm_v2/MMLU-PRO": 0.4494 } }, { diff --git a/data/developers/daemontatox.json b/data/developers/daemontatox.json index b8e6a28a7fbf1f38ce6fe03abcdad2859f985a0e..303c645fdab066808be1ab6783e6eea08f827b82 100644 --- a/data/developers/daemontatox.json +++ b/data/developers/daemontatox.json @@ -35,12 +35,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4383, - "hfopenllm_v2/BBH": 0.5034, - "hfopenllm_v2/MATH Level 5": 0.1443, + "hfopenllm_v2/IFEval": 0.4398, + "hfopenllm_v2/BBH": 0.5066, + "hfopenllm_v2/MATH Level 5": 0.1488, "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4052, - "hfopenllm_v2/MMLU-PRO": 0.3778 + "hfopenllm_v2/MUSR": 0.4079, + "hfopenllm_v2/MMLU-PRO": 0.3804 } }, { @@ -119,12 +119,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5064, - "hfopenllm_v2/BBH": 0.5112, - "hfopenllm_v2/MATH Level 5": 0.1631, - "hfopenllm_v2/GPQA": 0.3163, - "hfopenllm_v2/MUSR": 0.3973, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.777, + "hfopenllm_v2/BBH": 0.5187, + "hfopenllm_v2/MATH Level 5": 0.2198, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.3911, + "hfopenllm_v2/MMLU-PRO": 0.3738 } }, { diff --git a/data/developers/davielion.json b/data/developers/davielion.json index 5027340a9fce3c0b5162453c9d10516ec45234d2..d2a3ca1471ea06bca721c18d505591908b22252e 100644 --- a/data/developers/davielion.json +++ b/data/developers/davielion.json @@ -7,12 +7,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { diff --git a/data/developers/deepmount00.json b/data/developers/deepmount00.json index e898074e4a61782e05002f7b47eb2ee0411313aa..5505c28134c7ac10ef3a27978dd2745253c793a1 100644 --- a/data/developers/deepmount00.json +++ b/data/developers/deepmount00.json @@ -63,12 +63,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5365, - "hfopenllm_v2/BBH": 0.517, - "hfopenllm_v2/MATH Level 5": 0.1707, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4487, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.7917, + "hfopenllm_v2/BBH": 0.5109, + "hfopenllm_v2/MATH Level 5": 0.1088, + "hfopenllm_v2/GPQA": 0.2878, + "hfopenllm_v2/MUSR": 0.4136, + "hfopenllm_v2/MMLU-PRO": 0.3876 } }, { diff --git a/data/developers/epistemeai.json b/data/developers/epistemeai.json index ae59684b2ae8edda61f6033386c80ae35c1570fc..11b25cd31eead1ba78e7fde536703493137e208b 100644 --- a/data/developers/epistemeai.json +++ b/data/developers/epistemeai.json @@ -231,12 +231,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7305, - "hfopenllm_v2/BBH": 0.4649, - "hfopenllm_v2/MATH Level 5": 0.1397, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3209, - "hfopenllm_v2/MMLU-PRO": 0.348 + "hfopenllm_v2/IFEval": 0.7207, + "hfopenllm_v2/BBH": 0.461, + "hfopenllm_v2/MATH Level 5": 0.1314, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3432, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { diff --git a/data/developers/etherll.json b/data/developers/etherll.json index 6a72dd37a279f4d76a8244a057714c910854bcd2..2be2455f8558b72fcbd342c72cc671c443f3155e 100644 --- a/data/developers/etherll.json +++ b/data/developers/etherll.json @@ -35,12 +35,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6106, - "hfopenllm_v2/BBH": 0.5347, - "hfopenllm_v2/MATH Level 5": 0.1548, - "hfopenllm_v2/GPQA": 0.3146, - "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.3752 + "hfopenllm_v2/IFEval": 0.4672, + "hfopenllm_v2/BBH": 0.5013, + "hfopenllm_v2/MATH Level 5": 0.0279, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.386, + "hfopenllm_v2/MMLU-PRO": 0.3482 } }, { diff --git a/data/developers/goekdeniz-guelmez.json b/data/developers/goekdeniz-guelmez.json index c6e66fe9b3a75e2d49b70b9c90b10752326b37c4..e903240eb11157a57a20b31274250b901066286d 100644 --- a/data/developers/goekdeniz-guelmez.json +++ b/data/developers/goekdeniz-guelmez.json @@ -49,12 +49,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7628, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2802, - "hfopenllm_v2/MUSR": 0.4579, - "hfopenllm_v2/MMLU-PRO": 0.4033 + "hfopenllm_v2/IFEval": 0.7598, + "hfopenllm_v2/BBH": 0.5107, + "hfopenllm_v2/MATH Level 5": 0.4237, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.4539, + "hfopenllm_v2/MMLU-PRO": 0.4012 } }, { @@ -63,12 +63,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3417, - "hfopenllm_v2/BBH": 0.3292, - "hfopenllm_v2/MATH Level 5": 0.0023, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3249, - "hfopenllm_v2/MMLU-PRO": 0.1638 + "hfopenllm_v2/IFEval": 0.3472, + "hfopenllm_v2/BBH": 0.3268, + "hfopenllm_v2/MATH Level 5": 0.0891, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3262, + "hfopenllm_v2/MMLU-PRO": 0.1641 } }, { diff --git a/data/developers/google.json b/data/developers/google.json index 81f1df0599e79d5fa5b9eb113ec92dededc4447e..785568af1a47d9f2c7836777b5810f3a3f2fb1bd 100644 --- a/data/developers/google.json +++ b/data/developers/google.json @@ -157,6 +157,8 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Overall Score": 0.47, + "ace/Gaming Score": 0.509, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.373, "apex-agents/Overall Mean Score": 0.341, @@ -164,8 +166,6 @@ "apex-agents/Management Consulting Pass@1": 0.124, "apex-agents/Corporate Law Pass@1": 0.239, "apex-agents/Corporate Lawyer Mean Score": 0.487, - "ace/Overall Score": 0.47, - "ace/Gaming Score": 0.509, "apex-v1/Overall Score": 0.643, "apex-v1/Consulting Score": 0.64, "apex-v1/Investment Banking Score": 0.63 @@ -723,7 +723,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 15.4 + "terminal-bench-2.0/terminal-bench-2.0": 17.1 } }, { @@ -861,7 +861,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 51.0 + "terminal-bench-2.0/terminal-bench-2.0": 64.3 } }, { @@ -870,7 +870,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.2 + "terminal-bench-2.0/terminal-bench-2.0": 61.8 } }, { @@ -901,9 +901,9 @@ "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, - "tau-bench-2_retail/tau-bench-2/retail": 0.82, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 + "tau-bench-2_airline/tau-bench-2/airline": 0.62, + "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 } }, { @@ -912,7 +912,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.8 + "terminal-bench-2.0/terminal-bench-2.0": 78.4 } }, { @@ -1028,12 +1028,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1993, - "hfopenllm_v2/BBH": 0.3656, - "hfopenllm_v2/MATH Level 5": 0.0287, + "hfopenllm_v2/IFEval": 0.2018, + "hfopenllm_v2/BBH": 0.3709, + "hfopenllm_v2/MATH Level 5": 0.0302, "hfopenllm_v2/GPQA": 0.2626, - "hfopenllm_v2/MUSR": 0.4232, - "hfopenllm_v2/MMLU-PRO": 0.218 + "hfopenllm_v2/MUSR": 0.4219, + "hfopenllm_v2/MMLU-PRO": 0.2217 } }, { diff --git a/data/developers/gunulhona.json b/data/developers/gunulhona.json index 3d63c85d80df83c6632e31f815d8dc78510d19a0..1eba4dc6aed35ea93d90c3c8c2e1a2676805ffa0 100644 --- a/data/developers/gunulhona.json +++ b/data/developers/gunulhona.json @@ -21,12 +21,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } } ] diff --git a/data/developers/huggingfacetb.json b/data/developers/huggingfacetb.json index bed31781473fb30427be579aab45ef01bf5054ce..60e6e856fd10ebf29a5d127dd18301cd9edf10dd 100644 --- a/data/developers/huggingfacetb.json +++ b/data/developers/huggingfacetb.json @@ -161,12 +161,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } } ] diff --git a/data/developers/internlm.json b/data/developers/internlm.json index 035007101b6dacaf9414107420239eb159a0ec86..fbcc7249ad65cfb92e1e09809d6765ef54a982e2 100644 --- a/data/developers/internlm.json +++ b/data/developers/internlm.json @@ -21,16 +21,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8217, + "reward-bench/Score": 0.3902, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Safety": 0.4711, + "reward-bench/Reasoning": 0.8724, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, - "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Reasoning": 0.8724 + "reward-bench/Ties": 0.1934 } }, { diff --git a/data/developers/jaspionjader.json b/data/developers/jaspionjader.json index 9d9d1e268e56a9945ae657deca0493de6a22ce3d..053d128582b4aa02040ae51cd0577288669f17e5 100644 --- a/data/developers/jaspionjader.json +++ b/data/developers/jaspionjader.json @@ -1477,12 +1477,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4418, - "hfopenllm_v2/BBH": 0.5406, - "hfopenllm_v2/MATH Level 5": 0.1352, - "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5419, + "hfopenllm_v2/MATH Level 5": 0.1292, + "hfopenllm_v2/GPQA": 0.3087, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.386 + "hfopenllm_v2/MMLU-PRO": 0.3854 } }, { diff --git a/data/developers/leroydyer.json b/data/developers/leroydyer.json index e1aa95462e1fed8d4e011735c7846112031a21a1..119d29bab457804b8d64637d339b1f72d7389de3 100644 --- a/data/developers/leroydyer.json +++ b/data/developers/leroydyer.json @@ -707,12 +707,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3579, - "hfopenllm_v2/BBH": 0.4477, - "hfopenllm_v2/MATH Level 5": 0.0423, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4134, - "hfopenllm_v2/MMLU-PRO": 0.2376 + "hfopenllm_v2/IFEval": 0.3798, + "hfopenllm_v2/BBH": 0.4483, + "hfopenllm_v2/MATH Level 5": 0.04, + "hfopenllm_v2/GPQA": 0.3129, + "hfopenllm_v2/MUSR": 0.4148, + "hfopenllm_v2/MMLU-PRO": 0.2389 } }, { diff --git a/data/developers/llmat.json b/data/developers/llmat.json index 95633d3199310803501c256677a0fb788d0a08f7..d073eb81547c22e04e5363702192b3a9654d7362 100644 --- a/data/developers/llmat.json +++ b/data/developers/llmat.json @@ -7,12 +7,12 @@ "developer": "llmat", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.364, - "hfopenllm_v2/BBH": 0.4005, - "hfopenllm_v2/MATH Level 5": 0.0015, - "hfopenllm_v2/GPQA": 0.2693, - "hfopenllm_v2/MUSR": 0.3529, - "hfopenllm_v2/MMLU-PRO": 0.2301 + "hfopenllm_v2/IFEval": 0.377, + "hfopenllm_v2/BBH": 0.3978, + "hfopenllm_v2/MATH Level 5": 0.0242, + "hfopenllm_v2/GPQA": 0.2668, + "hfopenllm_v2/MUSR": 0.3555, + "hfopenllm_v2/MMLU-PRO": 0.2278 } } ] diff --git a/data/developers/magpie-align.json b/data/developers/magpie-align.json index 0dc0a43e89bb456a61006caa30051add55effb08..155a416e715db19186e6af681f6736ea9ed101d1 100644 --- a/data/developers/magpie-align.json +++ b/data/developers/magpie-align.json @@ -35,12 +35,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4027, - "hfopenllm_v2/BBH": 0.4789, - "hfopenllm_v2/MATH Level 5": 0.0461, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.3087, - "hfopenllm_v2/MMLU-PRO": 0.3001 + "hfopenllm_v2/IFEval": 0.4118, + "hfopenllm_v2/BBH": 0.4811, + "hfopenllm_v2/MATH Level 5": 0.034, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3047, + "hfopenllm_v2/MMLU-PRO": 0.3006 } }, { diff --git a/data/developers/meta-llama.json b/data/developers/meta-llama.json index 76d923e16aab37465ed7dda94d7be71e39f4b29e..a28b7c096e670398e8b74d7b30002bc56044221b 100644 --- a/data/developers/meta-llama.json +++ b/data/developers/meta-llama.json @@ -265,12 +265,12 @@ "developer": "meta-llama", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7408, - "hfopenllm_v2/BBH": 0.4989, - "hfopenllm_v2/MATH Level 5": 0.0869, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3568, - "hfopenllm_v2/MMLU-PRO": 0.3664, + "hfopenllm_v2/IFEval": 0.4782, + "hfopenllm_v2/BBH": 0.491, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2928, + "hfopenllm_v2/MUSR": 0.3805, + "hfopenllm_v2/MMLU-PRO": 0.3591, "reward-bench/Score": 0.645, "reward-bench/Chat": 0.8547, "reward-bench/Chat Hard": 0.4156, diff --git a/data/developers/microsoft.json b/data/developers/microsoft.json index 5b22c3abc624b7d65a825afffb8ded406b5eb203..7ea0eb813b21f5c278a71169ee533064bbab2740 100644 --- a/data/developers/microsoft.json +++ b/data/developers/microsoft.json @@ -225,12 +225,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5477, - "hfopenllm_v2/BBH": 0.5491, - "hfopenllm_v2/MATH Level 5": 0.1639, - "hfopenllm_v2/GPQA": 0.3322, - "hfopenllm_v2/MUSR": 0.4284, - "hfopenllm_v2/MMLU-PRO": 0.4022 + "hfopenllm_v2/IFEval": 0.5613, + "hfopenllm_v2/BBH": 0.5676, + "hfopenllm_v2/MATH Level 5": 0.1163, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.395, + "hfopenllm_v2/MMLU-PRO": 0.3866 } }, { diff --git a/data/developers/minimax.json b/data/developers/minimax.json index b575a16ccd7fb272ceb4b3067b0a9e48f65cff08..3eb98fb6a6609e5f1cc4d74a47ecc2b74aaa9bb0 100644 --- a/data/developers/minimax.json +++ b/data/developers/minimax.json @@ -25,7 +25,7 @@ "developer": "MiniMax", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 29.2 + "terminal-bench-2.0/terminal-bench-2.0": 36.6 } }, { diff --git a/data/developers/moonshot_ai.json b/data/developers/moonshot_ai.json index 746185ce773a539bd025922ab856fdcf2f8a1d9f..d83f11402fcca0c39c33d09eed495f2aefd69384 100644 --- a/data/developers/moonshot_ai.json +++ b/data/developers/moonshot_ai.json @@ -7,7 +7,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.8 + "terminal-bench-2.0/terminal-bench-2.0": 26.7 } }, { diff --git a/data/developers/nazimali.json b/data/developers/nazimali.json index 07c42beb48b2fc9e0ad1a1059178a9ce4071a591..34d47c9647d462b17a0fe6015c5c4f9fc00264e7 100644 --- a/data/developers/nazimali.json +++ b/data/developers/nazimali.json @@ -21,12 +21,12 @@ "developer": "nazimali", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4964, - "hfopenllm_v2/BBH": 0.4699, - "hfopenllm_v2/MATH Level 5": 0.0045, - "hfopenllm_v2/GPQA": 0.2827, - "hfopenllm_v2/MUSR": 0.3979, - "hfopenllm_v2/MMLU-PRO": 0.3063 + "hfopenllm_v2/IFEval": 0.486, + "hfopenllm_v2/BBH": 0.4721, + "hfopenllm_v2/MATH Level 5": 0.0846, + "hfopenllm_v2/GPQA": 0.2844, + "hfopenllm_v2/MUSR": 0.4006, + "hfopenllm_v2/MMLU-PRO": 0.3087 } } ] diff --git a/data/developers/nicolinho.json b/data/developers/nicolinho.json index bf9706fe4ed328188b0945860efeebf7183abc65..79bf445ae201aa8b9add0559d92e4abd4fd3bebb 100644 --- a/data/developers/nicolinho.json +++ b/data/developers/nicolinho.json @@ -7,16 +7,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9444, + "reward-bench/Score": 0.7667, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Safety": 0.9578, + "reward-bench/Reasoning": 0.9826, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Reasoning": 0.9826 + "reward-bench/Ties": 0.8321 } }, { diff --git a/data/developers/omkar1102.json b/data/developers/omkar1102.json index ca0270b469e58068ee5242b6801e2fbe782ddc0a..1d044781189744d770af993cfdb651c1e02eee6f 100644 --- a/data/developers/omkar1102.json +++ b/data/developers/omkar1102.json @@ -7,12 +7,12 @@ "developer": "Omkar1102", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2148, - "hfopenllm_v2/BBH": 0.276, + "hfopenllm_v2/IFEval": 0.2254, + "hfopenllm_v2/BBH": 0.275, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2508, - "hfopenllm_v2/MUSR": 0.3802, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3762, + "hfopenllm_v2/MMLU-PRO": 0.1123 } } ] diff --git a/data/developers/ontocord.json b/data/developers/ontocord.json index c16f4cbacdda3485b721f459b079923a6793a670..add41dbe183800ebed90c916e0b704c974dd50e7 100644 --- a/data/developers/ontocord.json +++ b/data/developers/ontocord.json @@ -273,12 +273,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1128, - "hfopenllm_v2/BBH": 0.3171, - "hfopenllm_v2/MATH Level 5": 0.0113, - "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.346, - "hfopenllm_v2/MMLU-PRO": 0.1129 + "hfopenllm_v2/IFEval": 0.1162, + "hfopenllm_v2/BBH": 0.3184, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2634, + "hfopenllm_v2/MUSR": 0.3447, + "hfopenllm_v2/MMLU-PRO": 0.1124 } }, { diff --git a/data/developers/openai.json b/data/developers/openai.json index 778495ca5d9933e50afc2a386628c3370e3c9523..7ffb402ef626c167784b6adff2263ee1471b3836 100644 --- a/data/developers/openai.json +++ b/data/developers/openai.json @@ -859,16 +859,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.5796, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8374, + "reward-bench/Score": 0.8007, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962 + "reward-bench/Ties": 0.6962, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Reasoning": 0.8374 } }, { @@ -877,7 +877,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 49.6 + "terminal-bench-2.0/terminal-bench-2.0": 35.2 } }, { @@ -922,7 +922,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 43.4 + "terminal-bench-2.0/terminal-bench-2.0": 44.3 } }, { @@ -931,7 +931,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 34.8 + "terminal-bench-2.0/terminal-bench-2.0": 24.0 } }, { @@ -954,7 +954,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 + "terminal-bench-2.0/terminal-bench-2.0": 7.9 } }, { @@ -1013,7 +1013,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.7 + "terminal-bench-2.0/terminal-bench-2.0": 54.0 } }, { @@ -1022,15 +1022,15 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.071, - "browsecompplus/browsecompplus": 0.46, + "appworld_test_normal/appworld/test_normal": 0.0, + "browsecompplus/browsecompplus": 0.43, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.57, + "swe-bench/swe-bench": 0.5455, "tau-bench-2_airline/tau-bench-2/airline": 0.6, "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, { @@ -1233,9 +1233,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.014084507042253521, - "livecodebenchpro/Medium Problems": 0.30985915492957744, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0143, + "livecodebenchpro/Medium Problems": 0.2923, + "livecodebenchpro/Easy Problems": 0.8571 } }, { diff --git a/data/developers/openbmb.json b/data/developers/openbmb.json index d8dae84054074ce01b5c47fc58b69a148fdc99c0..dd2ac9b2651a2fee615e466e3b3da81ac3cfeef4 100644 --- a/data/developers/openbmb.json +++ b/data/developers/openbmb.json @@ -21,17 +21,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8159, + "reward-bench/Score": 0.5806, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Safety": 0.6267, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, - "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172 + "reward-bench/Ties": 0.5972 } }, { @@ -68,17 +68,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4683, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Safety": 0.5089, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294, + "reward-bench/Score": 0.6903, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, + "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036 + "reward-bench/Ties": 0.3036, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294 } } ] diff --git a/data/developers/pku-alignment.json b/data/developers/pku-alignment.json index b42d9be156feba13e7cdbffbd62351a64e28c1fb..4cebe3d4c55c06b2bbbd44b7bb6fe487a030b922 100644 --- a/data/developers/pku-alignment.json +++ b/data/developers/pku-alignment.json @@ -7,17 +7,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5798, + "reward-bench/Score": 0.3332, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Safety": 0.7589, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, - "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57 + "reward-bench/Ties": -0.01 } }, { diff --git a/data/developers/primeintellect.json b/data/developers/primeintellect.json index 160722785b06f80d9b220dee435fad1245d45495..674a0e3b141480d7e0d33d0a2fe9b205b710216f 100644 --- a/data/developers/primeintellect.json +++ b/data/developers/primeintellect.json @@ -8,11 +8,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.274, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.25, - "hfopenllm_v2/MUSR": 0.3753, - "hfopenllm_v2/MMLU-PRO": 0.112 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3339, + "hfopenllm_v2/MMLU-PRO": 0.1123 } }, { diff --git a/data/developers/qingy2019.json b/data/developers/qingy2019.json index 3885f54f7f780ebe87cee5b8aacc1c1136f4441f..9e607d380bbf6837078a684f50c178b1c83bab9f 100644 --- a/data/developers/qingy2019.json +++ b/data/developers/qingy2019.json @@ -35,12 +35,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2358, - "hfopenllm_v2/BBH": 0.4612, - "hfopenllm_v2/MATH Level 5": 0.0642, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3717, - "hfopenllm_v2/MMLU-PRO": 0.2382 + "hfopenllm_v2/IFEval": 0.2401, + "hfopenllm_v2/BBH": 0.4622, + "hfopenllm_v2/MATH Level 5": 0.0725, + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3703, + "hfopenllm_v2/MMLU-PRO": 0.2379 } }, { @@ -49,12 +49,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6066, - "hfopenllm_v2/BBH": 0.635, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.3725, + "hfopenllm_v2/IFEval": 0.6005, + "hfopenllm_v2/BBH": 0.6356, + "hfopenllm_v2/MATH Level 5": 0.2764, + "hfopenllm_v2/GPQA": 0.3691, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5331 + "hfopenllm_v2/MMLU-PRO": 0.5339 } }, { diff --git a/data/developers/ray2333.json b/data/developers/ray2333.json index 709d24161c8237750555868d00eabe667376cddb..e188da7e1b361f275f3618471481e7b52afe105b 100644 --- a/data/developers/ray2333.json +++ b/data/developers/ray2333.json @@ -79,17 +79,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.589, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209, + "reward-bench/Score": 0.8464, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743 + "reward-bench/Ties": 0.5743, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209 } }, { diff --git a/data/developers/replete-ai.json b/data/developers/replete-ai.json index 0f038b31b0c0cf28a02b6c25fb4fd9bd374c118c..dbb06f00736a7fcddf76b24a5c7673e098ecab57 100644 --- a/data/developers/replete-ai.json +++ b/data/developers/replete-ai.json @@ -91,12 +91,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { diff --git a/data/developers/riaz.json b/data/developers/riaz.json index 342d5654379425c9866753c909eb0413c18e5c77..d54f5bbaf70d322fa875b56275f987dd3ae70f34 100644 --- a/data/developers/riaz.json +++ b/data/developers/riaz.json @@ -7,12 +7,12 @@ "developer": "riaz", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4137, - "hfopenllm_v2/BBH": 0.4565, - "hfopenllm_v2/MATH Level 5": 0.0453, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3776, - "hfopenllm_v2/MMLU-PRO": 0.2978 + "hfopenllm_v2/IFEval": 0.4373, + "hfopenllm_v2/BBH": 0.4586, + "hfopenllm_v2/MATH Level 5": 0.0514, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3763, + "hfopenllm_v2/MMLU-PRO": 0.2964 } } ] diff --git a/data/developers/sao10k.json b/data/developers/sao10k.json index e66900f1906266fdf37e55995d3b04130f5b6fd2..6569058a39b468de833ea3ab9419e073af211e5a 100644 --- a/data/developers/sao10k.json +++ b/data/developers/sao10k.json @@ -35,12 +35,12 @@ "developer": "Sao10K", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7384, - "hfopenllm_v2/BBH": 0.6471, - "hfopenllm_v2/MATH Level 5": 0.2137, + "hfopenllm_v2/IFEval": 0.7281, + "hfopenllm_v2/BBH": 0.6503, + "hfopenllm_v2/MATH Level 5": 0.2243, "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4209, - "hfopenllm_v2/MMLU-PRO": 0.5104 + "hfopenllm_v2/MUSR": 0.4196, + "hfopenllm_v2/MMLU-PRO": 0.5096 } }, { diff --git a/data/developers/sfairxc.json b/data/developers/sfairxc.json index 0f83a5aa819fcf69906563e63fd5c5a7cc1f8ce0..f511504fa472f754b653d8fff6432038f2fa5642 100644 --- a/data/developers/sfairxc.json +++ b/data/developers/sfairxc.json @@ -7,17 +7,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8338, + "reward-bench/Score": 0.6292, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492 + "reward-bench/Ties": 0.6647 } } ] diff --git a/data/developers/skywork.json b/data/developers/skywork.json index b75ac4fa3e86dd1c341887c14306e39e694944d2..310fc474921f3ff87a2e73d5f74077c21e25d2fe 100644 --- a/data/developers/skywork.json +++ b/data/developers/skywork.json @@ -47,16 +47,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.938, + "reward-bench/Score": 0.7576, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Safety": 0.9422, + "reward-bench/Reasoning": 0.9606, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, - "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Reasoning": 0.9606 + "reward-bench/Ties": 0.8261 } }, { @@ -89,16 +89,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7314, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8728, - "reward-bench/Safety": 0.9333, - "reward-bench/Reasoning": 0.962, + "reward-bench/Score": 0.9252, "reward-bench/Factuality": 0.6989, "reward-bench/Precise IF": 0.425, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.9616, - "reward-bench/Ties": 0.741 + "reward-bench/Ties": 0.741, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8728, + "reward-bench/Reasoning": 0.962 } }, { @@ -230,16 +230,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9007, + "reward-bench/Score": 0.6885, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.875, + "reward-bench/Safety": 0.8911, + "reward-bench/Reasoning": 0.9176, "reward-bench/Factuality": 0.6063, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6339, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7586, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.875, - "reward-bench/Reasoning": 0.9176 + "reward-bench/Ties": 0.7586 } } ] diff --git a/data/developers/sometimesanotion.json b/data/developers/sometimesanotion.json index c020e85fc5c690d215cbd09f94a114e50c6a6936..b756bb971c60b6bcad2e85989edb63b8a758fe59 100644 --- a/data/developers/sometimesanotion.json +++ b/data/developers/sometimesanotion.json @@ -749,12 +749,12 @@ "developer": "sometimesanotion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5278, - "hfopenllm_v2/BBH": 0.6557, - "hfopenllm_v2/MATH Level 5": 0.3119, - "hfopenllm_v2/GPQA": 0.3842, - "hfopenllm_v2/MUSR": 0.4754, - "hfopenllm_v2/MMLU-PRO": 0.5396 + "hfopenllm_v2/IFEval": 0.5367, + "hfopenllm_v2/BBH": 0.6561, + "hfopenllm_v2/MATH Level 5": 0.358, + "hfopenllm_v2/GPQA": 0.3867, + "hfopenllm_v2/MUSR": 0.474, + "hfopenllm_v2/MMLU-PRO": 0.5395 } }, { diff --git a/data/developers/spow12.json b/data/developers/spow12.json index 39b5b7b70e78310868166d1a8d2f2ae80dc77b8f..fbe5ef8e79c87a1486a23e55dcad74bc6e42909c 100644 --- a/data/developers/spow12.json +++ b/data/developers/spow12.json @@ -49,12 +49,12 @@ "developer": "spow12", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6511, - "hfopenllm_v2/BBH": 0.5926, - "hfopenllm_v2/MATH Level 5": 0.1858, - "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/IFEval": 0.6517, + "hfopenllm_v2/BBH": 0.5908, + "hfopenllm_v2/MATH Level 5": 0.2032, + "hfopenllm_v2/GPQA": 0.3238, "hfopenllm_v2/MUSR": 0.3842, - "hfopenllm_v2/MMLU-PRO": 0.3836 + "hfopenllm_v2/MMLU-PRO": 0.3812 } } ] diff --git a/data/developers/tanliboy.json b/data/developers/tanliboy.json index 7b17e651b33310ae087a6456408950e1456ff623..daa17945601d8528cbfc7c883e4dc4315ea363fc 100644 --- a/data/developers/tanliboy.json +++ b/data/developers/tanliboy.json @@ -7,12 +7,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1829, - "hfopenllm_v2/BBH": 0.5488, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.4056, - "hfopenllm_v2/MMLU-PRO": 0.3805 + "hfopenllm_v2/IFEval": 0.4501, + "hfopenllm_v2/BBH": 0.5472, + "hfopenllm_v2/MATH Level 5": 0.0944, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4017, + "hfopenllm_v2/MMLU-PRO": 0.3792 } }, { diff --git a/data/developers/valiantlabs.json b/data/developers/valiantlabs.json index a0fc2b3d81b7ddfe31ae4c63f4f370e45c708501..d692f13110a559faa51989d7851579fce9b6dc8e 100644 --- a/data/developers/valiantlabs.json +++ b/data/developers/valiantlabs.json @@ -49,12 +49,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7168, - "hfopenllm_v2/BBH": 0.4911, - "hfopenllm_v2/MATH Level 5": 0.1533, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.3512, - "hfopenllm_v2/MMLU-PRO": 0.3663 + "hfopenllm_v2/IFEval": 0.3496, + "hfopenllm_v2/BBH": 0.4947, + "hfopenllm_v2/MATH Level 5": 0.1269, + "hfopenllm_v2/GPQA": 0.3037, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.3644 } }, { @@ -105,12 +105,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2678, - "hfopenllm_v2/BBH": 0.4429, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.302, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.2927 + "hfopenllm_v2/IFEval": 0.6496, + "hfopenllm_v2/BBH": 0.4774, + "hfopenllm_v2/MATH Level 5": 0.0566, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.3909, + "hfopenllm_v2/MMLU-PRO": 0.3382 } }, { diff --git a/data/developers/weqweasdas.json b/data/developers/weqweasdas.json index e11faa49d0109edbe3a144d975a897042bc51bca..53701e6ba9018d055fac2463f132ffbb9e4ba26c 100644 --- a/data/developers/weqweasdas.json +++ b/data/developers/weqweasdas.json @@ -7,17 +7,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5027, + "reward-bench/Score": 0.2498, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.24, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, - "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564 + "reward-bench/Ties": 0.0315 } }, { @@ -26,17 +26,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3057, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.4079, - "reward-bench/Safety": 0.3311, - "reward-bench/Reasoning": 0.7637, - "reward-bench/Prior Sets (0.5 weight)": 0.6652, + "reward-bench/Score": 0.6549, "reward-bench/Factuality": 0.3705, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.4317, + "reward-bench/Safety": 0.4986, "reward-bench/Focus": 0.2343, - "reward-bench/Ties": 0.1851 + "reward-bench/Ties": 0.1851, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.4079, + "reward-bench/Reasoning": 0.7637, + "reward-bench/Prior Sets (0.5 weight)": 0.6652 } }, { @@ -78,17 +78,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.596, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Safety": 0.6911, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753, + "reward-bench/Score": 0.7982, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, + "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226 + "reward-bench/Ties": 0.6226, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753 } } ] diff --git a/data/developers/xai.json b/data/developers/xai.json index 4539ab8505af399ecb6f06dee4d4f2618de3e3bc..d0ba30a6d3c372a4bdb6c48d3c0cb90677ed30d5 100644 --- a/data/developers/xai.json +++ b/data/developers/xai.json @@ -120,7 +120,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 25.8 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } } ] diff --git a/data/developers/ycros.json b/data/developers/ycros.json index e95400442aba80bbdb2ffa85224e7399ce946b0d..9b83f7f872f197f98fd71538cda202af618f5659 100644 --- a/data/developers/ycros.json +++ b/data/developers/ycros.json @@ -7,12 +7,12 @@ "developer": "ycros", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6262, - "hfopenllm_v2/BBH": 0.5142, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3079, - "hfopenllm_v2/MUSR": 0.4138, - "hfopenllm_v2/MMLU-PRO": 0.3481 + "hfopenllm_v2/IFEval": 0.5994, + "hfopenllm_v2/BBH": 0.5159, + "hfopenllm_v2/MATH Level 5": 0.0785, + "hfopenllm_v2/GPQA": 0.3045, + "hfopenllm_v2/MUSR": 0.4203, + "hfopenllm_v2/MMLU-PRO": 0.3473 } } ] diff --git a/data/developers/yoyo-ai.json b/data/developers/yoyo-ai.json index d0e307257898b20d6b2c21faa52d04a28d7cca51..9dcbc81e86094f67bbf3f24e1917205fb4cc44f8 100644 --- a/data/developers/yoyo-ai.json +++ b/data/developers/yoyo-ai.json @@ -105,12 +105,12 @@ "developer": "YOYO-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7905, - "hfopenllm_v2/BBH": 0.6406, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3163, - "hfopenllm_v2/MUSR": 0.4181, - "hfopenllm_v2/MMLU-PRO": 0.4944 + "hfopenllm_v2/IFEval": 0.5899, + "hfopenllm_v2/BBH": 0.654, + "hfopenllm_v2/MATH Level 5": 0.4509, + "hfopenllm_v2/GPQA": 0.3834, + "hfopenllm_v2/MUSR": 0.4744, + "hfopenllm_v2/MMLU-PRO": 0.5376 } }, { diff --git a/data/models.json b/data/models.json index a5af0007574aa89e984bf7d8a4b2911e04bfd788..12fa873edd89b9fe064fc8ee02af3da9906fdc49 100644 --- a/data/models.json +++ b/data/models.json @@ -1391,10 +1391,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6924, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.3575, - "reward-bench/Safety": 0.7757 + "reward-bench/Score": 0.6895, + "reward-bench/Chat": 0.9385, + "reward-bench/Chat Hard": 0.3706, + "reward-bench/Safety": 0.7595 } }, { @@ -2036,12 +2036,12 @@ "developer": "akjindal53244", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8051, - "hfopenllm_v2/BBH": 0.5189, - "hfopenllm_v2/MATH Level 5": 0.1722, - "hfopenllm_v2/GPQA": 0.3263, + "hfopenllm_v2/IFEval": 0.8033, + "hfopenllm_v2/BBH": 0.5196, + "hfopenllm_v2/MATH Level 5": 0.1624, + "hfopenllm_v2/GPQA": 0.3096, "hfopenllm_v2/MUSR": 0.4028, - "hfopenllm_v2/MMLU-PRO": 0.3803 + "hfopenllm_v2/MMLU-PRO": 0.3812 } }, { @@ -2243,7 +2243,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.9 + "terminal-bench-2.0/terminal-bench-2.0": 27.2 } }, { @@ -2489,17 +2489,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.722, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Safety": 0.8689, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8892, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, + "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308 + "reward-bench/Ties": 0.8308, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -2508,12 +2508,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { @@ -2536,17 +2536,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8431, + "reward-bench/Score": 0.687, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.86, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6397 } }, { @@ -6556,11 +6556,11 @@ "developer": "amd", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1842, - "hfopenllm_v2/BBH": 0.2974, - "hfopenllm_v2/MATH Level 5": 0.0053, - "hfopenllm_v2/GPQA": 0.2525, - "hfopenllm_v2/MUSR": 0.378, + "hfopenllm_v2/IFEval": 0.1918, + "hfopenllm_v2/BBH": 0.2969, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/MUSR": 0.3846, "hfopenllm_v2/MMLU-PRO": 0.1169 } }, @@ -7167,17 +7167,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.3711, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Safety": 0.595, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635, + "reward-bench/Score": 0.7289, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, + "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899 + "reward-bench/Ties": 0.0899, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635 } }, { @@ -7232,16 +7232,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.8008, + "reward-bench/Score": 0.5744, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.8378, + "reward-bench/Reasoning": 0.7868, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Reasoning": 0.7868 + "reward-bench/Ties": 0.5601 } }, { @@ -7321,7 +7321,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 13.9 + "terminal-bench-2.0/terminal-bench-2.0": 28.3 } }, { @@ -7446,11 +7446,11 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.66, + "appworld_test_normal/appworld/test_normal": 0.68, "browsecompplus/browsecompplus": 0.49, - "swe-bench/swe-bench": 0.65, - "tau-bench-2_airline/tau-bench-2/airline": 0.72, - "tau-bench-2_retail/tau-bench-2/retail": 0.85, + "swe-bench/swe-bench": 0.8072, + "tau-bench-2_airline/tau-bench-2/airline": 0.66, + "tau-bench-2_retail/tau-bench-2/retail": 0.83, "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 } }, @@ -7460,7 +7460,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -7478,7 +7478,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.7 + "terminal-bench-2.0/terminal-bench-2.0": 62.9 } }, { @@ -7730,12 +7730,12 @@ "developer": "arcee-ai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5718, - "hfopenllm_v2/BBH": 0.5481, - "hfopenllm_v2/MATH Level 5": 0.114, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4008, - "hfopenllm_v2/MMLU-PRO": 0.3813 + "hfopenllm_v2/IFEval": 0.5621, + "hfopenllm_v2/BBH": 0.5489, + "hfopenllm_v2/MATH Level 5": 0.2953, + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.4021, + "hfopenllm_v2/MMLU-PRO": 0.3822 } }, { @@ -10599,12 +10599,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1775, - "hfopenllm_v2/BBH": 0.295, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1049 + "hfopenllm_v2/IFEval": 0.4652, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.1284, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3394, + "hfopenllm_v2/MMLU-PRO": 0.3152 } }, { @@ -11828,17 +11828,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } }, { @@ -12127,12 +12127,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4124, - "hfopenllm_v2/BBH": 0.6383, - "hfopenllm_v2/MATH Level 5": 0.182, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4349, - "hfopenllm_v2/MMLU-PRO": 0.4525 + "hfopenllm_v2/IFEval": 0.3613, + "hfopenllm_v2/BBH": 0.6123, + "hfopenllm_v2/MATH Level 5": 0.1239, + "hfopenllm_v2/GPQA": 0.328, + "hfopenllm_v2/MUSR": 0.4112, + "hfopenllm_v2/MMLU-PRO": 0.4494 } }, { @@ -14142,12 +14142,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4383, - "hfopenllm_v2/BBH": 0.5034, - "hfopenllm_v2/MATH Level 5": 0.1443, + "hfopenllm_v2/IFEval": 0.4398, + "hfopenllm_v2/BBH": 0.5066, + "hfopenllm_v2/MATH Level 5": 0.1488, "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4052, - "hfopenllm_v2/MMLU-PRO": 0.3778 + "hfopenllm_v2/MUSR": 0.4079, + "hfopenllm_v2/MMLU-PRO": 0.3804 } }, { @@ -14226,12 +14226,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5064, - "hfopenllm_v2/BBH": 0.5112, - "hfopenllm_v2/MATH Level 5": 0.1631, - "hfopenllm_v2/GPQA": 0.3163, - "hfopenllm_v2/MUSR": 0.3973, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.777, + "hfopenllm_v2/BBH": 0.5187, + "hfopenllm_v2/MATH Level 5": 0.2198, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.3911, + "hfopenllm_v2/MMLU-PRO": 0.3738 } }, { @@ -15393,12 +15393,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { @@ -15729,12 +15729,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5365, - "hfopenllm_v2/BBH": 0.517, - "hfopenllm_v2/MATH Level 5": 0.1707, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4487, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.7917, + "hfopenllm_v2/BBH": 0.5109, + "hfopenllm_v2/MATH Level 5": 0.1088, + "hfopenllm_v2/GPQA": 0.2878, + "hfopenllm_v2/MUSR": 0.4136, + "hfopenllm_v2/MMLU-PRO": 0.3876 } }, { @@ -20340,12 +20340,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7305, - "hfopenllm_v2/BBH": 0.4649, - "hfopenllm_v2/MATH Level 5": 0.1397, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3209, - "hfopenllm_v2/MMLU-PRO": 0.348 + "hfopenllm_v2/IFEval": 0.7207, + "hfopenllm_v2/BBH": 0.461, + "hfopenllm_v2/MATH Level 5": 0.1314, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3432, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { @@ -21040,12 +21040,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6106, - "hfopenllm_v2/BBH": 0.5347, - "hfopenllm_v2/MATH Level 5": 0.1548, - "hfopenllm_v2/GPQA": 0.3146, - "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.3752 + "hfopenllm_v2/IFEval": 0.4672, + "hfopenllm_v2/BBH": 0.5013, + "hfopenllm_v2/MATH Level 5": 0.0279, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.386, + "hfopenllm_v2/MMLU-PRO": 0.3482 } }, { @@ -23289,12 +23289,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7628, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2802, - "hfopenllm_v2/MUSR": 0.4579, - "hfopenllm_v2/MMLU-PRO": 0.4033 + "hfopenllm_v2/IFEval": 0.7598, + "hfopenllm_v2/BBH": 0.5107, + "hfopenllm_v2/MATH Level 5": 0.4237, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.4539, + "hfopenllm_v2/MMLU-PRO": 0.4012 } }, { @@ -23303,12 +23303,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3417, - "hfopenllm_v2/BBH": 0.3292, - "hfopenllm_v2/MATH Level 5": 0.0023, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3249, - "hfopenllm_v2/MMLU-PRO": 0.1638 + "hfopenllm_v2/IFEval": 0.3472, + "hfopenllm_v2/BBH": 0.3268, + "hfopenllm_v2/MATH Level 5": 0.0891, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3262, + "hfopenllm_v2/MMLU-PRO": 0.1641 } }, { @@ -23537,6 +23537,8 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Overall Score": 0.47, + "ace/Gaming Score": 0.509, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.373, "apex-agents/Overall Mean Score": 0.341, @@ -23544,8 +23546,6 @@ "apex-agents/Management Consulting Pass@1": 0.124, "apex-agents/Corporate Law Pass@1": 0.239, "apex-agents/Corporate Lawyer Mean Score": 0.487, - "ace/Overall Score": 0.47, - "ace/Gaming Score": 0.509, "apex-v1/Overall Score": 0.643, "apex-v1/Consulting Score": 0.64, "apex-v1/Investment Banking Score": 0.63 @@ -24103,7 +24103,7 @@ "reward-bench/Safety": 0.909, "reward-bench/Focus": 0.841, "reward-bench/Ties": 0.809, - "terminal-bench-2.0/terminal-bench-2.0": 15.4 + "terminal-bench-2.0/terminal-bench-2.0": 17.1 } }, { @@ -24241,7 +24241,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 51.0 + "terminal-bench-2.0/terminal-bench-2.0": 64.3 } }, { @@ -24250,7 +24250,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.2 + "terminal-bench-2.0/terminal-bench-2.0": 61.8 } }, { @@ -24281,9 +24281,9 @@ "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, - "tau-bench-2_retail/tau-bench-2/retail": 0.82, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 + "tau-bench-2_airline/tau-bench-2/airline": 0.62, + "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.6852 } }, { @@ -24292,7 +24292,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 74.8 + "terminal-bench-2.0/terminal-bench-2.0": 78.4 } }, { @@ -24408,12 +24408,12 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1993, - "hfopenllm_v2/BBH": 0.3656, - "hfopenllm_v2/MATH Level 5": 0.0287, + "hfopenllm_v2/IFEval": 0.2018, + "hfopenllm_v2/BBH": 0.3709, + "hfopenllm_v2/MATH Level 5": 0.0302, "hfopenllm_v2/GPQA": 0.2626, - "hfopenllm_v2/MUSR": 0.4232, - "hfopenllm_v2/MMLU-PRO": 0.218 + "hfopenllm_v2/MUSR": 0.4219, + "hfopenllm_v2/MMLU-PRO": 0.2217 } }, { @@ -25558,12 +25558,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } }, { @@ -26814,12 +26814,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } }, { @@ -28651,16 +28651,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8217, + "reward-bench/Score": 0.3902, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Safety": 0.4711, + "reward-bench/Reasoning": 0.8724, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, - "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Reasoning": 0.8724 + "reward-bench/Ties": 0.1934 } }, { @@ -30623,12 +30623,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4418, - "hfopenllm_v2/BBH": 0.5406, - "hfopenllm_v2/MATH Level 5": 0.1352, - "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5419, + "hfopenllm_v2/MATH Level 5": 0.1292, + "hfopenllm_v2/GPQA": 0.3087, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.386 + "hfopenllm_v2/MMLU-PRO": 0.3854 } }, { @@ -38012,12 +38012,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3579, - "hfopenllm_v2/BBH": 0.4477, - "hfopenllm_v2/MATH Level 5": 0.0423, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4134, - "hfopenllm_v2/MMLU-PRO": 0.2376 + "hfopenllm_v2/IFEval": 0.3798, + "hfopenllm_v2/BBH": 0.4483, + "hfopenllm_v2/MATH Level 5": 0.04, + "hfopenllm_v2/GPQA": 0.3129, + "hfopenllm_v2/MUSR": 0.4148, + "hfopenllm_v2/MMLU-PRO": 0.2389 } }, { @@ -38936,12 +38936,12 @@ "developer": "llmat", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.364, - "hfopenllm_v2/BBH": 0.4005, - "hfopenllm_v2/MATH Level 5": 0.0015, - "hfopenllm_v2/GPQA": 0.2693, - "hfopenllm_v2/MUSR": 0.3529, - "hfopenllm_v2/MMLU-PRO": 0.2301 + "hfopenllm_v2/IFEval": 0.377, + "hfopenllm_v2/BBH": 0.3978, + "hfopenllm_v2/MATH Level 5": 0.0242, + "hfopenllm_v2/GPQA": 0.2668, + "hfopenllm_v2/MUSR": 0.3555, + "hfopenllm_v2/MMLU-PRO": 0.2278 } }, { @@ -39741,12 +39741,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4027, - "hfopenllm_v2/BBH": 0.4789, - "hfopenllm_v2/MATH Level 5": 0.0461, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.3087, - "hfopenllm_v2/MMLU-PRO": 0.3001 + "hfopenllm_v2/IFEval": 0.4118, + "hfopenllm_v2/BBH": 0.4811, + "hfopenllm_v2/MATH Level 5": 0.034, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3047, + "hfopenllm_v2/MMLU-PRO": 0.3006 } }, { @@ -42056,12 +42056,12 @@ "developer": "meta-llama", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7408, - "hfopenllm_v2/BBH": 0.4989, - "hfopenllm_v2/MATH Level 5": 0.0869, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3568, - "hfopenllm_v2/MMLU-PRO": 0.3664, + "hfopenllm_v2/IFEval": 0.4782, + "hfopenllm_v2/BBH": 0.491, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2928, + "hfopenllm_v2/MUSR": 0.3805, + "hfopenllm_v2/MMLU-PRO": 0.3591, "reward-bench/Score": 0.645, "reward-bench/Chat": 0.8547, "reward-bench/Chat Hard": 0.4156, @@ -43235,12 +43235,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5477, - "hfopenllm_v2/BBH": 0.5491, - "hfopenllm_v2/MATH Level 5": 0.1639, - "hfopenllm_v2/GPQA": 0.3322, - "hfopenllm_v2/MUSR": 0.4284, - "hfopenllm_v2/MMLU-PRO": 0.4022 + "hfopenllm_v2/IFEval": 0.5613, + "hfopenllm_v2/BBH": 0.5676, + "hfopenllm_v2/MATH Level 5": 0.1163, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.395, + "hfopenllm_v2/MMLU-PRO": 0.3866 } }, { @@ -43667,7 +43667,7 @@ "developer": "MiniMax", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 29.2 + "terminal-bench-2.0/terminal-bench-2.0": 36.6 } }, { @@ -44996,7 +44996,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.8 + "terminal-bench-2.0/terminal-bench-2.0": 26.7 } }, { @@ -45633,12 +45633,12 @@ "developer": "nazimali", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4964, - "hfopenllm_v2/BBH": 0.4699, - "hfopenllm_v2/MATH Level 5": 0.0045, - "hfopenllm_v2/GPQA": 0.2827, - "hfopenllm_v2/MUSR": 0.3979, - "hfopenllm_v2/MMLU-PRO": 0.3063 + "hfopenllm_v2/IFEval": 0.486, + "hfopenllm_v2/BBH": 0.4721, + "hfopenllm_v2/MATH Level 5": 0.0846, + "hfopenllm_v2/GPQA": 0.2844, + "hfopenllm_v2/MUSR": 0.4006, + "hfopenllm_v2/MMLU-PRO": 0.3087 } }, { @@ -48273,16 +48273,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9444, + "reward-bench/Score": 0.7667, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Safety": 0.9578, + "reward-bench/Reasoning": 0.9826, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Reasoning": 0.9826 + "reward-bench/Ties": 0.8321 } }, { @@ -50085,12 +50085,12 @@ "developer": "Omkar1102", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2148, - "hfopenllm_v2/BBH": 0.276, + "hfopenllm_v2/IFEval": 0.2254, + "hfopenllm_v2/BBH": 0.275, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2508, - "hfopenllm_v2/MUSR": 0.3802, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3762, + "hfopenllm_v2/MMLU-PRO": 0.1123 } }, { @@ -50393,12 +50393,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1128, - "hfopenllm_v2/BBH": 0.3171, - "hfopenllm_v2/MATH Level 5": 0.0113, - "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.346, - "hfopenllm_v2/MMLU-PRO": 0.1129 + "hfopenllm_v2/IFEval": 0.1162, + "hfopenllm_v2/BBH": 0.3184, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2634, + "hfopenllm_v2/MUSR": 0.3447, + "hfopenllm_v2/MMLU-PRO": 0.1124 } }, { @@ -51707,16 +51707,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.5796, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Safety": 0.7667, - "reward-bench/Reasoning": 0.8374, + "reward-bench/Score": 0.8007, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962 + "reward-bench/Ties": 0.6962, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Reasoning": 0.8374 } }, { @@ -51725,7 +51725,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 49.6 + "terminal-bench-2.0/terminal-bench-2.0": 35.2 } }, { @@ -51770,7 +51770,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 43.4 + "terminal-bench-2.0/terminal-bench-2.0": 44.3 } }, { @@ -51779,7 +51779,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 34.8 + "terminal-bench-2.0/terminal-bench-2.0": 24.0 } }, { @@ -51802,7 +51802,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 + "terminal-bench-2.0/terminal-bench-2.0": 7.9 } }, { @@ -51861,7 +51861,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.7 + "terminal-bench-2.0/terminal-bench-2.0": 54.0 } }, { @@ -51870,15 +51870,15 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.071, - "browsecompplus/browsecompplus": 0.46, + "appworld_test_normal/appworld/test_normal": 0.0, + "browsecompplus/browsecompplus": 0.43, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.57, + "swe-bench/swe-bench": 0.5455, "tau-bench-2_airline/tau-bench-2/airline": 0.6, "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 + "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, { @@ -52081,9 +52081,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.014084507042253521, - "livecodebenchpro/Medium Problems": 0.30985915492957744, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0143, + "livecodebenchpro/Medium Problems": 0.2923, + "livecodebenchpro/Easy Problems": 0.8571 } }, { @@ -52312,17 +52312,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8159, + "reward-bench/Score": 0.5806, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Safety": 0.6267, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, - "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172 + "reward-bench/Ties": 0.5972 } }, { @@ -52359,17 +52359,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4683, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Safety": 0.5089, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294, + "reward-bench/Score": 0.6903, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, + "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036 + "reward-bench/Ties": 0.3036, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294 } }, { @@ -53830,17 +53830,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5798, + "reward-bench/Score": 0.3332, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Safety": 0.7589, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, - "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57 + "reward-bench/Ties": -0.01 } }, { @@ -54172,11 +54172,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.274, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.25, - "hfopenllm_v2/MUSR": 0.3753, - "hfopenllm_v2/MMLU-PRO": 0.112 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3339, + "hfopenllm_v2/MMLU-PRO": 0.1123 } }, { @@ -56591,12 +56591,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2358, - "hfopenllm_v2/BBH": 0.4612, - "hfopenllm_v2/MATH Level 5": 0.0642, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3717, - "hfopenllm_v2/MMLU-PRO": 0.2382 + "hfopenllm_v2/IFEval": 0.2401, + "hfopenllm_v2/BBH": 0.4622, + "hfopenllm_v2/MATH Level 5": 0.0725, + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3703, + "hfopenllm_v2/MMLU-PRO": 0.2379 } }, { @@ -56605,12 +56605,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6066, - "hfopenllm_v2/BBH": 0.635, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.3725, + "hfopenllm_v2/IFEval": 0.6005, + "hfopenllm_v2/BBH": 0.6356, + "hfopenllm_v2/MATH Level 5": 0.2764, + "hfopenllm_v2/GPQA": 0.3691, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5331 + "hfopenllm_v2/MMLU-PRO": 0.5339 } }, { @@ -59400,17 +59400,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.589, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.6842, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9133, - "reward-bench/Prior Sets (0.5 weight)": 0.7209, + "reward-bench/Score": 0.8464, "reward-bench/Factuality": 0.5874, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.6727, - "reward-bench/Ties": 0.5743 + "reward-bench/Ties": 0.5743, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.6842, + "reward-bench/Reasoning": 0.9133, + "reward-bench/Prior Sets (0.5 weight)": 0.7209 } }, { @@ -59721,12 +59721,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { @@ -59861,12 +59861,12 @@ "developer": "riaz", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4137, - "hfopenllm_v2/BBH": 0.4565, - "hfopenllm_v2/MATH Level 5": 0.0453, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3776, - "hfopenllm_v2/MMLU-PRO": 0.2978 + "hfopenllm_v2/IFEval": 0.4373, + "hfopenllm_v2/BBH": 0.4586, + "hfopenllm_v2/MATH Level 5": 0.0514, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3763, + "hfopenllm_v2/MMLU-PRO": 0.2964 } }, { @@ -61835,12 +61835,12 @@ "developer": "Sao10K", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7384, - "hfopenllm_v2/BBH": 0.6471, - "hfopenllm_v2/MATH Level 5": 0.2137, + "hfopenllm_v2/IFEval": 0.7281, + "hfopenllm_v2/BBH": 0.6503, + "hfopenllm_v2/MATH Level 5": 0.2243, "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4209, - "hfopenllm_v2/MMLU-PRO": 0.5104 + "hfopenllm_v2/MUSR": 0.4196, + "hfopenllm_v2/MMLU-PRO": 0.5096 } }, { @@ -62492,17 +62492,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8338, + "reward-bench/Score": 0.6292, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492 + "reward-bench/Ties": 0.6647 } }, { @@ -63283,16 +63283,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.938, + "reward-bench/Score": 0.7576, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Safety": 0.9422, + "reward-bench/Reasoning": 0.9606, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, - "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Reasoning": 0.9606 + "reward-bench/Ties": 0.8261 } }, { @@ -63325,16 +63325,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7314, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8728, - "reward-bench/Safety": 0.9333, - "reward-bench/Reasoning": 0.962, + "reward-bench/Score": 0.9252, "reward-bench/Factuality": 0.6989, "reward-bench/Precise IF": 0.425, "reward-bench/Math": 0.6284, + "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.9616, - "reward-bench/Ties": 0.741 + "reward-bench/Ties": 0.741, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8728, + "reward-bench/Reasoning": 0.962 } }, { @@ -63466,16 +63466,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9007, + "reward-bench/Score": 0.6885, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.875, + "reward-bench/Safety": 0.8911, + "reward-bench/Reasoning": 0.9176, "reward-bench/Factuality": 0.6063, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6339, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7586, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.875, - "reward-bench/Reasoning": 0.9176 + "reward-bench/Ties": 0.7586 } }, { @@ -64322,12 +64322,12 @@ "developer": "sometimesanotion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5278, - "hfopenllm_v2/BBH": 0.6557, - "hfopenllm_v2/MATH Level 5": 0.3119, - "hfopenllm_v2/GPQA": 0.3842, - "hfopenllm_v2/MUSR": 0.4754, - "hfopenllm_v2/MMLU-PRO": 0.5396 + "hfopenllm_v2/IFEval": 0.5367, + "hfopenllm_v2/BBH": 0.6561, + "hfopenllm_v2/MATH Level 5": 0.358, + "hfopenllm_v2/GPQA": 0.3867, + "hfopenllm_v2/MUSR": 0.474, + "hfopenllm_v2/MMLU-PRO": 0.5395 } }, { @@ -64728,12 +64728,12 @@ "developer": "spow12", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6511, - "hfopenllm_v2/BBH": 0.5926, - "hfopenllm_v2/MATH Level 5": 0.1858, - "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/IFEval": 0.6517, + "hfopenllm_v2/BBH": 0.5908, + "hfopenllm_v2/MATH Level 5": 0.2032, + "hfopenllm_v2/GPQA": 0.3238, "hfopenllm_v2/MUSR": 0.3842, - "hfopenllm_v2/MMLU-PRO": 0.3836 + "hfopenllm_v2/MMLU-PRO": 0.3812 } }, { @@ -66784,12 +66784,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1829, - "hfopenllm_v2/BBH": 0.5488, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.4056, - "hfopenllm_v2/MMLU-PRO": 0.3805 + "hfopenllm_v2/IFEval": 0.4501, + "hfopenllm_v2/BBH": 0.5472, + "hfopenllm_v2/MATH Level 5": 0.0944, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4017, + "hfopenllm_v2/MMLU-PRO": 0.3792 } }, { @@ -71000,12 +71000,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7168, - "hfopenllm_v2/BBH": 0.4911, - "hfopenllm_v2/MATH Level 5": 0.1533, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.3512, - "hfopenllm_v2/MMLU-PRO": 0.3663 + "hfopenllm_v2/IFEval": 0.3496, + "hfopenllm_v2/BBH": 0.4947, + "hfopenllm_v2/MATH Level 5": 0.1269, + "hfopenllm_v2/GPQA": 0.3037, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.3644 } }, { @@ -71056,12 +71056,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2678, - "hfopenllm_v2/BBH": 0.4429, - "hfopenllm_v2/MATH Level 5": 0.0521, - "hfopenllm_v2/GPQA": 0.302, - "hfopenllm_v2/MUSR": 0.3959, - "hfopenllm_v2/MMLU-PRO": 0.2927 + "hfopenllm_v2/IFEval": 0.6496, + "hfopenllm_v2/BBH": 0.4774, + "hfopenllm_v2/MATH Level 5": 0.0566, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.3909, + "hfopenllm_v2/MMLU-PRO": 0.3382 } }, { @@ -71686,17 +71686,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5027, + "reward-bench/Score": 0.2498, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.24, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, - "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564 + "reward-bench/Ties": 0.0315 } }, { @@ -71705,17 +71705,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3057, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.4079, - "reward-bench/Safety": 0.3311, - "reward-bench/Reasoning": 0.7637, - "reward-bench/Prior Sets (0.5 weight)": 0.6652, + "reward-bench/Score": 0.6549, "reward-bench/Factuality": 0.3705, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.4317, + "reward-bench/Safety": 0.4986, "reward-bench/Focus": 0.2343, - "reward-bench/Ties": 0.1851 + "reward-bench/Ties": 0.1851, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.4079, + "reward-bench/Reasoning": 0.7637, + "reward-bench/Prior Sets (0.5 weight)": 0.6652 } }, { @@ -71757,17 +71757,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.596, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Safety": 0.6911, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753, + "reward-bench/Score": 0.7982, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, + "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226 + "reward-bench/Ties": 0.6226, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753 } }, { @@ -72436,7 +72436,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 25.8 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } }, { @@ -73140,12 +73140,12 @@ "developer": "ycros", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6262, - "hfopenllm_v2/BBH": 0.5142, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3079, - "hfopenllm_v2/MUSR": 0.4138, - "hfopenllm_v2/MMLU-PRO": 0.3481 + "hfopenllm_v2/IFEval": 0.5994, + "hfopenllm_v2/BBH": 0.5159, + "hfopenllm_v2/MATH Level 5": 0.0785, + "hfopenllm_v2/GPQA": 0.3045, + "hfopenllm_v2/MUSR": 0.4203, + "hfopenllm_v2/MMLU-PRO": 0.3473 } }, { @@ -73826,12 +73826,12 @@ "developer": "YOYO-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7905, - "hfopenllm_v2/BBH": 0.6406, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3163, - "hfopenllm_v2/MUSR": 0.4181, - "hfopenllm_v2/MMLU-PRO": 0.4944 + "hfopenllm_v2/IFEval": 0.5899, + "hfopenllm_v2/BBH": 0.654, + "hfopenllm_v2/MATH Level 5": 0.4509, + "hfopenllm_v2/GPQA": 0.3834, + "hfopenllm_v2/MUSR": 0.4744, + "hfopenllm_v2/MMLU-PRO": 0.5376 } }, { diff --git a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json index e9d556b28d7550c1e76a8e996c98b852c7622823..036d2144a17c2773cc45b3fcdc429ec4c85d29b1 100644 --- a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json +++ b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json @@ -38,7 +38,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6905 + "score": 0.7058 }, "source_data": { "dataset_name": "RewardBench", @@ -56,7 +56,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9525 }, "source_data": { "dataset_name": "RewardBench", @@ -74,7 +74,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3947 }, "source_data": { "dataset_name": "RewardBench", @@ -92,7 +92,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7676 + "score": 0.7703 }, "source_data": { "dataset_name": "RewardBench", @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6808 + "score": 0.6905 }, "source_data": { "dataset_name": "RewardBench", @@ -152,7 +152,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -188,7 +188,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7527 + "score": 0.7676 }, "source_data": { "dataset_name": "RewardBench", @@ -230,7 +230,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6895 + "score": 0.7004 }, "source_data": { "dataset_name": "RewardBench", @@ -248,7 +248,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9413 }, "source_data": { "dataset_name": "RewardBench", @@ -266,7 +266,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -284,7 +284,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7595 + "score": 0.7716 }, "source_data": { "dataset_name": "RewardBench", @@ -326,7 +326,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7004 + "score": 0.6945 }, "source_data": { "dataset_name": "RewardBench", @@ -344,7 +344,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9413 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -362,7 +362,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -380,7 +380,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7716 + "score": 0.7743 }, "source_data": { "dataset_name": "RewardBench", @@ -422,7 +422,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7019 + "score": 0.6808 }, "source_data": { "dataset_name": "RewardBench", @@ -440,7 +440,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.9302 }, "source_data": { "dataset_name": "RewardBench", @@ -458,7 +458,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -476,7 +476,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7811 + "score": 0.7527 }, "source_data": { "dataset_name": "RewardBench", @@ -518,7 +518,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7058 + "score": 0.7008 }, "source_data": { "dataset_name": "RewardBench", @@ -536,7 +536,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9525 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -554,7 +554,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3947 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -572,7 +572,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7703 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -614,7 +614,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7008 + "score": 0.7019 }, "source_data": { "dataset_name": "RewardBench", @@ -632,7 +632,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9497 }, "source_data": { "dataset_name": "RewardBench", @@ -650,7 +650,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.375 }, "source_data": { "dataset_name": "RewardBench", @@ -668,7 +668,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7811 }, "source_data": { "dataset_name": "RewardBench", @@ -710,7 +710,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6945 + "score": 0.6924 }, "source_data": { "dataset_name": "RewardBench", @@ -728,7 +728,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -746,7 +746,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3575 }, "source_data": { "dataset_name": "RewardBench", @@ -764,7 +764,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7743 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -806,7 +806,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6924 + "score": 0.6895 }, "source_data": { "dataset_name": "RewardBench", @@ -824,7 +824,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -842,7 +842,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3575 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -860,7 +860,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7595 }, "source_data": { "dataset_name": "RewardBench", diff --git a/data/models/akjindal53244_llama-3.1-storm-8b.json b/data/models/akjindal53244_llama-3.1-storm-8b.json index 10080988b150f205c06f956e608326efa2dd3fb0..cd06cbbbbe2ccc306c529154ea621a16fb9ebcf6 100644 --- a/data/models/akjindal53244_llama-3.1-storm-8b.json +++ b/data/models/akjindal53244_llama-3.1-storm-8b.json @@ -5,7 +5,7 @@ "developer": "akjindal53244", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8033 + "score": 0.8051 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5196 + "score": 0.5189 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1624 + "score": 0.1722 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3263 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3812 + "score": 0.3803 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8051 + "score": 0.8033 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5189 + "score": 0.5196 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1722 + "score": 0.1624 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.3096 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3803 + "score": 0.3812 } } ], diff --git a/data/models/alibaba_qwen-3-coder-480b.json b/data/models/alibaba_qwen-3-coder-480b.json index 78f6ba994c4cff4ccaf572ddc5297bedb19efb9b..76736ecf57c8a847549ed4d201c4f8542bcb4b8e 100644 --- a/data/models/alibaba_qwen-3-coder-480b.json +++ b/data/models/alibaba_qwen-3-coder-480b.json @@ -4,13 +4,13 @@ "id": "alibaba/qwen-3-coder-480b", "developer": "Alibaba", "additional_details": { - "agent_name": "Dakou Agent", - "agent_organization": "iflow" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-28", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-12-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 23.9, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json index 3c470e6abd70f5f2bc9a8be57cc36f939ca7753c..56c7eedcfadd2447274047e1824ef99e14cf4e28 100644 --- a/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8892 + "score": 0.722 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.8084 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8268 + "score": 0.3688 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6776 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9027 + "score": 0.8689 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8583 + "score": 0.7778 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8308 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.722 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8084 + "score": 0.8892 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3688 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.8268 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8689 + "score": 0.9027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7778 + "score": 0.8583 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8308 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json index 52eed39924ec31b815aba5be343333e3d61411ae..408cdcea425aaae910c976c1f57587d29fcf9c63 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.687 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7516 + "score": 0.8431 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.86 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8545 + "score": 0.7898 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6397 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8431 + "score": 0.687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.7516 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.86 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7898 + "score": 0.8545 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.6397 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b.json b/data/models/allenai_llama-3.1-tulu-3-8b.json index 7de1c9431728784c04f1c32781631febc2ed7e32..53350f20b1f74556a0af6a9aa87ed77d181deaec 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8255 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.405 + "score": 0.4061 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1964 + "score": 0.2115 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2987 + "score": 0.297 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2821 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8255 + "score": 0.8267 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4061 + "score": 0.405 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2115 + "score": 0.1964 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.297 + "score": 0.2987 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2821 + "score": 0.2827 } } ], diff --git a/data/models/amd_amd-llama-135m.json b/data/models/amd_amd-llama-135m.json index dfdcf8e7a431ddc8bdf51b6aa797b81cf0f89c9e..a445333e3dd0495a4c3e2fddf3e92a96b288f85c 100644 --- a/data/models/amd_amd-llama-135m.json +++ b/data/models/amd_amd-llama-135m.json @@ -5,9 +5,9 @@ "developer": "amd", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", - "params_billions": "0.134" + "params_billions": "0.135" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1918 + "score": 0.1842 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2969 + "score": 0.2974 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0053 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2584 + "score": 0.2525 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3846 + "score": 0.378 } }, { @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1842 + "score": 0.1918 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2974 + "score": 0.2969 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0053 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2525 + "score": 0.2584 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.378 + "score": 0.3846 } }, { diff --git a/data/models/anthropic_claude-3-5-haiku-20241022.json b/data/models/anthropic_claude-3-5-haiku-20241022.json index 9edb0c7484cd3b1eab61fa90716227281f955253..97ab32cadbb5e110dfa409cbea1ee9b8c38d6850 100644 --- a/data/models/anthropic_claude-3-5-haiku-20241022.json +++ b/data/models/anthropic_claude-3-5-haiku-20241022.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-7-sonnet-20250219.json b/data/models/anthropic_claude-3-7-sonnet-20250219.json index 26744b8fcce13c56de7ede1b941e4f633dfc781a..6e02a0c482b2aed6381eb4f121bd334b0959f1b6 100644 --- a/data/models/anthropic_claude-3-7-sonnet-20250219.json +++ b/data/models/anthropic_claude-3-7-sonnet-20250219.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-haiku-20240307.json b/data/models/anthropic_claude-3-haiku-20240307.json index 52bb9959dd5613be4f3fe143bd6f01f97e2632c2..4a2b563f634d87e24d77a5003bd4befff9181e63 100644 --- a/data/models/anthropic_claude-3-haiku-20240307.json +++ b/data/models/anthropic_claude-3-haiku-20240307.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,109 +1925,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7289 + "score": 0.3711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9274 + "score": 0.4042 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5197 + "score": 0.2812 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3552 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7953 + "score": 0.595 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.706 + "score": 0.501 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6635 + "score": 0.0899 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -2035,10 +2053,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2057,127 +2075,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3711 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4042 + "score": 0.7289 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.9274 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3552 + "score": 0.5197 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.595 + "score": 0.7953 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.501 + "score": 0.706 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0899 + "score": 0.6635 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/anthropic_claude-3-opus-20240229.json b/data/models/anthropic_claude-3-opus-20240229.json index 150503f1a5f91571d10a33f8965d4bd84f3ccf46..8f8b70dd804fcbb225ceb8111209d93e20e8bd0b 100644 --- a/data/models/anthropic_claude-3-opus-20240229.json +++ b/data/models/anthropic_claude-3-opus-20240229.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", + "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,104 +1925,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5744 + "score": 0.8008 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5389 + "score": 0.9469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.6031 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5137 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8378 + "score": 0.7868 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6646 + "score": 0.5744 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2031,135 +2055,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5601 + "score": 0.5389 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8008 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9469 + "score": 0.5137 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.8378 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.6646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7868 + "score": 0.5601 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/anthropic_claude-haiku-4.5.json b/data/models/anthropic_claude-haiku-4.5.json index e7b7ed8c329f7f81d319b4442fafacedb7ba1133..5982de0c04db9149bd5fa6081ab9674c3bc855d7 100644 --- a/data/models/anthropic_claude-haiku-4.5.json +++ b/data/models/anthropic_claude-haiku-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-haiku-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Goose", + "agent_organization": "Block" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.8, + "score": 35.5, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 28.3, + "score": 13.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.5, + "score": 29.8, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 13.9, + "score": 28.3, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4-5.json b/data/models/anthropic_claude-opus-4-5.json index ec98797a0bada43df823aeb8285704dc256b7f56..47edf5cb77433abffb4d65f42865bc3c46d880ed 100644 --- a/data/models/anthropic_claude-opus-4-5.json +++ b/data/models/anthropic_claude-opus-4-5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4-5", "developer": "Anthropic", "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.64, + "score": 0.66, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.43", - "total_run_cost": "343.32", - "average_steps": "20.06", - "percent_finished": "0.82" + "average_agent_cost": "13.08", + "total_run_cost": "1308.38", + "average_steps": "49.69", + "percent_finished": "0.74" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -70,8 +70,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -282,7 +282,7 @@ } }, { - "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -295,42 +295,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5294, + "score": 0.64, "uncertainty": { - "num_samples": 51 + "num_samples": 100 }, "details": { - "average_agent_cost": "11.66", - "total_run_cost": "594.68", - "average_steps": "31.04", - "percent_finished": "0.8431" + "average_agent_cost": "3.43", + "total_run_cost": "343.32", + "average_steps": "20.06", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,23 +382,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.5294, "uncertainty": { - "num_samples": 100 + "num_samples": 51 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "11.66", + "total_run_cost": "594.68", + "average_steps": "31.04", + "percent_finished": "0.8431" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -431,34 +431,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.61, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "22.76", - "total_run_cost": "2276.48", - "average_steps": "47.65", - "percent_finished": "0.77" + "average_agent_cost": "7.59", + "total_run_cost": "759.44", + "average_steps": "27.18", + "percent_finished": "1.0" } }, "generation_config": { @@ -486,7 +486,7 @@ } }, { - "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.49, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.59", - "total_run_cost": "759.44", - "average_steps": "27.18", - "percent_finished": "1.0" + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "13.08", - "total_run_cost": "1308.38", - "average_steps": "49.69", - "percent_finished": "0.74" + "average_agent_cost": "22.76", + "total_run_cost": "2276.48", + "average_steps": "47.65", + "percent_finished": "0.77" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -682,15 +682,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -722,14 +722,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8072, + "score": 0.6061, "uncertainty": { - "num_samples": 83 + "num_samples": 99 }, "details": { - "average_agent_cost": "2.96", - "total_run_cost": "245.78", - "average_steps": "34.1", + "average_agent_cost": "3.97", + "total_run_cost": "393.16", + "average_steps": "43.44", "percent_finished": "1.0" } }, @@ -737,8 +737,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -750,15 +750,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -805,8 +805,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -818,8 +818,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -894,7 +894,7 @@ } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -926,14 +926,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6061, + "score": 0.65, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "3.97", - "total_run_cost": "393.16", - "average_steps": "43.44", + "average_agent_cost": "4.85", + "total_run_cost": "485.22", + "average_steps": "39.13", "percent_finished": "1.0" } }, @@ -941,8 +941,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -954,15 +954,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -994,14 +994,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.65, + "score": 0.8072, "uncertainty": { - "num_samples": 100 + "num_samples": 83 }, "details": { - "average_agent_cost": "4.85", - "total_run_cost": "485.22", - "average_steps": "39.13", + "average_agent_cost": "2.96", + "total_run_cost": "245.78", + "average_steps": "34.1", "percent_finished": "1.0" } }, @@ -1009,8 +1009,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1022,8 +1022,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1098,7 +1098,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1135,9 +1135,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "65.66", - "average_steps": "11.5", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1145,8 +1145,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1158,15 +1158,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1198,14 +1198,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.72, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "0.78", + "total_run_cost": "39.67", + "average_steps": "11.88", "percent_finished": "1.0" } }, @@ -1213,8 +1213,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1226,8 +1226,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1302,7 +1302,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1334,14 +1334,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.72, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.78", - "total_run_cost": "39.67", - "average_steps": "11.88", + "average_agent_cost": "1.3", + "total_run_cost": "65.66", + "average_steps": "11.5", "percent_finished": "1.0" } }, @@ -1349,8 +1349,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1362,8 +1362,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1438,7 +1438,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1470,13 +1470,13 @@ "max_score": 1.0 }, "score_details": { - "score": 0.83, + "score": 0.85, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.6", - "total_run_cost": "161.14", + "average_agent_cost": "0.55", + "total_run_cost": "56.18", "average_steps": "12.54", "percent_finished": "1.0" } @@ -1485,8 +1485,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1498,8 +1498,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1642,7 +1642,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1674,13 +1674,13 @@ "max_score": 1.0 }, "score_details": { - "score": 0.85, + "score": 0.83, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "56.18", + "average_agent_cost": "1.6", + "total_run_cost": "161.14", "average_steps": "12.54", "percent_finished": "1.0" } @@ -1689,8 +1689,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1702,8 +1702,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1778,7 +1778,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1815,9 +1815,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "2.45", + "total_run_cost": "255.97", + "average_steps": "18.71", "percent_finished": "1.0" } }, @@ -1825,8 +1825,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1838,15 +1838,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1883,9 +1883,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "2.45", - "total_run_cost": "255.97", - "average_steps": "18.71", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1893,8 +1893,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1906,8 +1906,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } diff --git a/data/models/anthropic_claude-opus-4.1.json b/data/models/anthropic_claude-opus-4.1.json index 643c986c59e6df513c07edafb084900951fa2ab2..f450f39387f472cdddd30c8581d35da9f459d5c6 100644 --- a/data/models/anthropic_claude-opus-4.1.json +++ b/data/models/anthropic_claude-opus-4.1.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.1", "developer": "Anthropic", "additional_details": { - "agent_name": "Claude Code", - "agent_organization": "Anthropic" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 35.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.1, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.5.json b/data/models/anthropic_claude-opus-4.5.json index 04c518e21852763638180858614bb6ac2be608de..09776a3455f602e0127758380687247631f384a0 100644 --- a/data/models/anthropic_claude-opus-4.5.json +++ b/data/models/anthropic_claude-opus-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "OpenCode", - "agent_organization": "Anomaly Innovations" + "agent_name": "Mux", + "agent_organization": "Coder" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-12", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,11 +43,11 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7 + "score": 58.4 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -64,7 +64,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -78,7 +78,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -102,7 +102,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-22", + "evaluation_timestamp": "2026-01-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -111,17 +111,11 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 51.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -138,7 +132,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -152,7 +146,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -176,7 +170,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -185,11 +179,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4 + "score": 63.1, + "uncertainty": { + "standard_error": { + "value": 2.7 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -206,7 +206,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -220,7 +220,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -244,7 +244,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -253,17 +253,17 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 54.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -280,7 +280,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -294,7 +294,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -318,7 +318,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -327,17 +327,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.3, + "score": 57.8, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -354,7 +354,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -368,7 +368,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -392,7 +392,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -401,17 +401,17 @@ "max_score": 100.0 }, "score_details": { - "score": 63.1, + "score": 52.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -428,7 +428,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -442,7 +442,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -466,7 +466,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -475,17 +475,17 @@ "max_score": 100.0 }, "score_details": { - "score": 52.1, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -502,7 +502,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.6.json b/data/models/anthropic_claude-opus-4.6.json index a41a27dbfdbf4960b47a33e1930b531d6b54f85b..cfa834e822d48dbd1b3a3875068947d83a36cefa 100644 --- a/data/models/anthropic_claude-opus-4.6.json +++ b/data/models/anthropic_claude-opus-4.6.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,11 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.9 + "score": 71.9, + "uncertainty": { + "standard_error": { + "value": 2.7 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -138,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-07", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,17 +265,11 @@ "max_score": 100.0 }, "score_details": { - "score": 58.0, - "uncertainty": { - "standard_error": { - "value": 2.9 - }, - "num_samples": 435 - } + "score": 66.9 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +300,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +324,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,17 +333,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, + "score": 58.0, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -407,17 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.9, + "score": 74.7, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.7, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-sonnet-4-20250514.json b/data/models/anthropic_claude-sonnet-4-20250514.json index 17485295f0940239338e7d9ae5edbc1db77fa9eb..a43572d10d77d034b8cc0b4e9e80cb595c19f907 100644 --- a/data/models/anthropic_claude-sonnet-4-20250514.json +++ b/data/models/anthropic_claude-sonnet-4-20250514.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-sonnet-4.5.json b/data/models/anthropic_claude-sonnet-4.5.json index 53dd8e2195a24f3d530bf69beec973cc6e94f1f4..ec3107a4e456d4dbbeb4419bd2e927231666a677 100644 --- a/data/models/anthropic_claude-sonnet-4.5.json +++ b/data/models/anthropic_claude-sonnet-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-sonnet-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "CAMEL-AI", - "agent_organization": "CAMEL-AI" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 46.5, + "score": 42.8, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.8, + "score": 46.5, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,11 @@ "max_score": 100.0 }, "score_details": { - "score": 42.5, - "uncertainty": { - "standard_error": { - "value": 2.8 - }, - "num_samples": 435 - } + "score": 42.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +300,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +324,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,11 +333,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.7 + "score": 43.1, + "uncertainty": { + "standard_error": { + "value": 2.6 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -398,7 +398,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -407,17 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.1, + "score": 42.5, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/arcee-ai_arcee-spark.json b/data/models/arcee-ai_arcee-spark.json index 611dbc409a44fd2eadd33b4e30a3048a3002308e..e194ce4efc974318b0bb01e7560254bdd047ab55 100644 --- a/data/models/arcee-ai_arcee-spark.json +++ b/data/models/arcee-ai_arcee-spark.json @@ -5,7 +5,7 @@ "developer": "arcee-ai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5621 + "score": 0.5718 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5489 + "score": 0.5481 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2953 + "score": 0.114 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3062 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4021 + "score": 0.4008 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3822 + "score": 0.3813 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5718 + "score": 0.5621 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5481 + "score": 0.5489 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.114 + "score": 0.2953 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.307 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4008 + "score": 0.4021 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3813 + "score": 0.3822 } } ], diff --git a/data/models/bunnycore_llama-3.2-3b-deep-test.json b/data/models/bunnycore_llama-3.2-3b-deep-test.json index 05cbb770cffa010b799551a630bc7b1089a91900..c829321bc4e1b62229f276602c62e6f6bd9a3faf 100644 --- a/data/models/bunnycore_llama-3.2-3b-deep-test.json +++ b/data/models/bunnycore_llama-3.2-3b-deep-test.json @@ -5,9 +5,9 @@ "developer": "bunnycore", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "3.607" + "params_billions": "1.803" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4652 + "score": 0.1775 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.295 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1284 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2643 + "score": 0.2517 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3394 + "score": 0.3647 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3152 + "score": 0.1049 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1775 + "score": 0.4652 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.295 + "score": 0.4531 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1284 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2643 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3647 + "score": 0.3394 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1049 + "score": 0.3152 } } ], diff --git a/data/models/cir-ams_btrm_qwen2_7b_0613.json b/data/models/cir-ams_btrm_qwen2_7b_0613.json index 9fb827ce32fb32044e2247d7f86c70d1bc13d414..84a36ab2262aa5d4869e5d141d383b85699971f1 100644 --- a/data/models/cir-ams_btrm_qwen2_7b_0613.json +++ b/data/models/cir-ams_btrm_qwen2_7b_0613.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5736 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.8172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3563 + "score": 0.9749 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7178 + "score": 0.9014 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5737 + "score": 0.8775 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6527 + "score": 0.7029 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8172 + "score": 0.5736 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9749 + "score": 0.5347 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3563 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6066 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9014 + "score": 0.7178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8775 + "score": 0.5737 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7029 + "score": 0.6527 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json b/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json index 9b7cc35f188c98dba3f0427de09e77d4b26cfc40..62d09c3db567da9932caf7141d7a1389c1aafea9 100644 --- a/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json +++ b/data/models/cognitivecomputations_dolphin-2.9.2-phi-3-medium-abliterated.json @@ -5,7 +5,7 @@ "developer": "cognitivecomputations", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "13.96" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3613 + "score": 0.4124 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6123 + "score": 0.6383 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1239 + "score": 0.182 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.328 + "score": 0.3289 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4112 + "score": 0.4349 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4494 + "score": 0.4525 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4124 + "score": 0.3613 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6383 + "score": 0.6123 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.182 + "score": 0.1239 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.328 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4349 + "score": 0.4112 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4525 + "score": 0.4494 } } ], diff --git a/data/models/cohere_command-a-03-2025.json b/data/models/cohere_command-a-03-2025.json index aedaf4e5f944a400d87afc543fa422bde24be135..205f7c496ed81b612dd2b182b3cf2e9fd9ac9c54 100644 --- a/data/models/cohere_command-a-03-2025.json +++ b/data/models/cohere_command-a-03-2025.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/daemontatox_aethertot.json b/data/models/daemontatox_aethertot.json index 36cfe9d7e08d8778379976a94c061a705e069a23..b20831e7b0add6cd1b35d7cacb70896114ddccd2 100644 --- a/data/models/daemontatox_aethertot.json +++ b/data/models/daemontatox_aethertot.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MllamaForConditionalGeneration", "params_billions": "10.67" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4398 + "score": 0.4383 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5066 + "score": 0.5034 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1488 + "score": 0.1443 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4079 + "score": 0.4052 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3804 + "score": 0.3778 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4383 + "score": 0.4398 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5034 + "score": 0.5066 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1443 + "score": 0.1488 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4052 + "score": 0.4079 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3778 + "score": 0.3804 } } ], diff --git a/data/models/daemontatox_documentcogito.json b/data/models/daemontatox_documentcogito.json index 6383820e3f19a97d2a08bca3546e5cca52a154e3..157a2466b6d71022e4fcfd8ace48804c5b301a99 100644 --- a/data/models/daemontatox_documentcogito.json +++ b/data/models/daemontatox_documentcogito.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MllamaForConditionalGeneration", "params_billions": "10.67" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.777 + "score": 0.5064 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5187 + "score": 0.5112 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2198 + "score": 0.1631 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2936 + "score": 0.3163 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3911 + "score": 0.3973 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3738 + "score": 0.3802 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5064 + "score": 0.777 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5112 + "score": 0.5187 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1631 + "score": 0.2198 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3163 + "score": 0.2936 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3973 + "score": 0.3911 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3738 } } ], diff --git a/data/models/davielion_llama-3.2-1b-spin-iter0.json b/data/models/davielion_llama-3.2-1b-spin-iter0.json index 849a1d52d59a5f16eab6aaa35f259f630d9dc175..2ea14b6e6da41d61b06880761473f45de63d9756 100644 --- a/data/models/davielion_llama-3.2-1b-spin-iter0.json +++ b/data/models/davielion_llama-3.2-1b-spin-iter0.json @@ -5,7 +5,7 @@ "developer": "DavieLion", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1549 + "score": 0.1507 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2937 + "score": 0.293 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.006 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2534 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1125 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1507 + "score": 0.1549 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.293 + "score": 0.2937 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.006 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2576 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1125 + "score": 0.1128 } } ], diff --git a/data/models/deepmount00_llama-3.1-8b-ita.json b/data/models/deepmount00_llama-3.1-8b-ita.json index 1fef7ca0b3d471692e7379016093b45312ecc7df..be94466036e753700b8de72e15b468fc6edebdee 100644 --- a/data/models/deepmount00_llama-3.1-8b-ita.json +++ b/data/models/deepmount00_llama-3.1-8b-ita.json @@ -6,8 +6,8 @@ "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", - "architecture": "LlamaForCausalLM", - "params_billions": "8.03", + "architecture": "Unknown", + "params_billions": "0.0", "model_id_aliases": [ "DeepMount00/Llama-3.1-8b-Ita" ] @@ -15,7 +15,7 @@ }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -47,7 +47,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7917 + "score": 0.5365 } }, { @@ -65,7 +65,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5109 + "score": 0.517 } }, { @@ -83,7 +83,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1088 + "score": 0.1707 } }, { @@ -101,7 +101,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2878 + "score": 0.3062 } }, { @@ -119,7 +119,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4136 + "score": 0.4487 } }, { @@ -137,7 +137,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3876 + "score": 0.396 } } ], @@ -145,7 +145,7 @@ "generation_config": null }, { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -177,7 +177,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5365 + "score": 0.7917 } }, { @@ -195,7 +195,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.517 + "score": 0.5109 } }, { @@ -213,7 +213,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1707 + "score": 0.1088 } }, { @@ -231,7 +231,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.2878 } }, { @@ -249,7 +249,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4487 + "score": 0.4136 } }, { @@ -267,7 +267,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.396 + "score": 0.3876 } } ], diff --git a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json index 623bd9c2c8bfaa513dd516b6dc8551699ef2135b..34df651aabdc1d7685c9419a18517c0c9ee2b5a2 100644 --- a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json +++ b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json @@ -5,7 +5,7 @@ "developer": "EpistemeAI", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7207 + "score": 0.7305 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.461 + "score": 0.4649 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1314 + "score": 0.1397 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3432 + "score": 0.3209 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.348 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7305 + "score": 0.7207 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4649 + "score": 0.461 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1397 + "score": 0.1314 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3209 + "score": 0.3432 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.348 + "score": 0.3354 } } ], diff --git a/data/models/etherll_herplete-llm-llama-3.1-8b.json b/data/models/etherll_herplete-llm-llama-3.1-8b.json index 1f50fb1261d1ec9c67c961ac513392f51c62e5ca..31e2291931d31d5d1deeb312e00cc5de025980fd 100644 --- a/data/models/etherll_herplete-llm-llama-3.1-8b.json +++ b/data/models/etherll_herplete-llm-llama-3.1-8b.json @@ -5,7 +5,7 @@ "developer": "Etherll", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4672 + "score": 0.6106 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5013 + "score": 0.5347 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0279 + "score": 0.1548 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3146 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3991 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3482 + "score": 0.3752 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6106 + "score": 0.4672 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.5013 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1548 + "score": 0.0279 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3146 + "score": 0.2861 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3991 + "score": 0.386 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3752 + "score": 0.3482 } } ], diff --git a/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json b/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json index 4037d46b51c1d5f1b2d23453c409340e1c080518..37f82f81803012f4270da262872645abffc35b8f 100644 --- a/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json +++ b/data/models/goekdeniz-guelmez_josie-7b-v6.0-step2000.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7598 + "score": 0.7628 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5107 + "score": 0.5098 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4237 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2802 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4539 + "score": 0.4579 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4012 + "score": 0.4033 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7628 + "score": 0.7598 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5107 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.4237 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2802 + "score": 0.2768 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4579 + "score": 0.4539 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4033 + "score": 0.4012 } } ], diff --git a/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json b/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json index 4e742d684d75a21064054d5a573ca557e7b1e590..57e8dc5e192d9afe5d8658b037afb0f25366b018 100644 --- a/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json +++ b/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "0.63" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3472 + "score": 0.3417 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3268 + "score": 0.3292 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0891 + "score": 0.0023 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2576 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3262 + "score": 0.3249 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1641 + "score": 0.1638 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3417 + "score": 0.3472 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3268 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0023 + "score": 0.0891 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2517 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3249 + "score": 0.3262 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1638 + "score": 0.1641 } } ], diff --git a/data/models/google_gemini-2.5-flash.json b/data/models/google_gemini-2.5-flash.json index 820c8d9942dde2ebcfddb6efd01e86ba89a1eddf..95ba532548298d2f931d1f8d4f2b4d963221f666 100644 --- a/data/models/google_gemini-2.5-flash.json +++ b/data/models/google_gemini-2.5-flash.json @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 17.1, + "score": 16.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,7 +1376,7 @@ "max_score": 100.0 }, "score_details": { - "score": 16.9, + "score": 16.4, "uncertainty": { "standard_error": { "value": 2.4 @@ -1386,7 +1386,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 15.4, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1491,7 +1491,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1515,7 +1515,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1524,17 +1524,17 @@ "max_score": 100.0 }, "score_details": { - "score": 15.4, + "score": 17.1, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1551,7 +1551,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-2.5-pro.json b/data/models/google_gemini-2.5-pro.json index ad69bc6f71351eff92acf181b72bb1ef3d2ca29a..64bf8e5b0bbd8066e28305bcfb70c0755706abd7 100644 --- a/data/models/google_gemini-2.5-pro.json +++ b/data/models/google_gemini-2.5-pro.json @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 32.6, + "score": 26.1, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.1, + "score": 19.6, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 19.6, + "score": 32.6, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-flash.json b/data/models/google_gemini-3-flash.json index 47fa2df5c9baf77e090796d7a217aa9e4577803b..16cea45f1dff16269cebc97fb714914d5b11c007 100644 --- a/data/models/google_gemini-3-flash.json +++ b/data/models/google_gemini-3-flash.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Gemini CLI", + "agent_organization": "Google" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-07", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7, + "score": 51.0, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.3, + "score": 51.7, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.0, + "score": 64.3, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-pro-preview.json b/data/models/google_gemini-3-pro-preview.json index 003d3891c3ed687a312bcf65a5ce96e8186ce448..95e6d99eeb9c9c27b3f039c22763603c38516fd7 100644 --- a/data/models/google_gemini-3-pro-preview.json +++ b/data/models/google_gemini-3-pro-preview.json @@ -78,7 +78,7 @@ } }, { - "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -91,42 +91,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.582, + "score": 0.57, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "8.7", - "total_run_cost": "869.55", - "average_steps": "33.49", - "percent_finished": "0.98" + "average_agent_cost": "2.39", + "total_run_cost": "239.0", + "average_steps": "29.63", + "percent_finished": "0.69" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -159,42 +159,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.85", - "total_run_cost": "284.68", - "average_steps": "22.88", - "percent_finished": "0.7" + "average_agent_cost": "1.3", + "total_run_cost": "130.49", + "average_steps": "22.59", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -227,42 +227,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "130.49", - "average_steps": "22.59", - "percent_finished": "1.0" + "average_agent_cost": "2.85", + "total_run_cost": "284.68", + "average_steps": "22.88", + "percent_finished": "0.7" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -295,42 +295,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.582, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.39", - "total_run_cost": "239.0", - "average_steps": "29.63", - "percent_finished": "0.69" + "average_agent_cost": "8.7", + "total_run_cost": "869.55", + "average_steps": "33.49", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -342,8 +342,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -554,7 +554,7 @@ } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -601,8 +601,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1720,7 +1720,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1752,14 +1752,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.67, + "score": 0.7576, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "3.68", - "total_run_cost": "367.97", - "average_steps": "43.72", + "average_agent_cost": "2.21", + "total_run_cost": "218.76", + "average_steps": "38.1", "percent_finished": "1.0" } }, @@ -1767,8 +1767,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1780,15 +1780,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1820,14 +1820,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.71, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "2.21", - "total_run_cost": "218.76", - "average_steps": "38.1", + "average_agent_cost": "0.7", + "total_run_cost": "69.56", + "average_steps": "32.55", "percent_finished": "1.0" } }, @@ -1835,8 +1835,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1848,15 +1848,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1903,8 +1903,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1916,15 +1916,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1956,14 +1956,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.67, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.7", - "total_run_cost": "69.56", - "average_steps": "32.55", + "average_agent_cost": "3.68", + "total_run_cost": "367.97", + "average_steps": "43.72", "percent_finished": "1.0" } }, @@ -1971,8 +1971,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1984,8 +1984,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2060,7 +2060,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2097,9 +2097,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.34", + "total_run_cost": "17.45", + "average_steps": "12.62", "percent_finished": "1.0" } }, @@ -2107,8 +2107,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2120,15 +2120,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2160,14 +2160,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.62, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.18", - "average_steps": "10.9", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2175,8 +2175,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2188,15 +2188,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2228,14 +2228,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.68, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.34", - "total_run_cost": "17.45", - "average_steps": "12.62", + "average_agent_cost": "0.2", + "total_run_cost": "10.29", + "average_steps": "12.28", "percent_finished": "1.0" } }, @@ -2243,8 +2243,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2256,8 +2256,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2332,7 +2332,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2364,14 +2364,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.62, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.2", - "total_run_cost": "10.29", - "average_steps": "12.28", + "average_agent_cost": "0.21", + "total_run_cost": "11.18", + "average_steps": "10.9", "percent_finished": "1.0" } }, @@ -2379,8 +2379,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2392,8 +2392,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2468,7 +2468,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2500,14 +2500,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7805, + "score": 0.82, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.19", - "total_run_cost": "19.38", - "average_steps": "11.18", + "average_agent_cost": "0.16", + "total_run_cost": "16.64", + "average_steps": "11.25", "percent_finished": "1.0" } }, @@ -2515,8 +2515,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2528,15 +2528,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2568,14 +2568,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.82, + "score": 0.7805, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "16.64", - "average_steps": "11.25", + "average_agent_cost": "0.19", + "total_run_cost": "19.38", + "average_steps": "11.18", "percent_finished": "1.0" } }, @@ -2583,8 +2583,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2596,15 +2596,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2636,14 +2636,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.82, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.27", - "total_run_cost": "27.48", - "average_steps": "10.62", + "average_agent_cost": "0.16", + "total_run_cost": "16.64", + "average_steps": "11.25", "percent_finished": "1.0" } }, @@ -2651,8 +2651,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2664,15 +2664,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2704,14 +2704,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.82, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "16.64", - "average_steps": "11.25", + "average_agent_cost": "0.27", + "total_run_cost": "27.48", + "average_steps": "10.62", "percent_finished": "1.0" } }, @@ -2719,8 +2719,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2732,15 +2732,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2772,14 +2772,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.88, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.35", + "total_run_cost": "40.25", + "average_steps": "12.71", "percent_finished": "1.0" } }, @@ -2787,8 +2787,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2800,15 +2800,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2840,23 +2840,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8876, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.54", - "total_run_cost": "58.29", - "average_steps": "10.82", - "percent_finished": "0.89" + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2868,15 +2868,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2908,23 +2908,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6852, + "score": 0.8876, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "25.48", - "average_steps": "9.9", - "percent_finished": "1.0" + "average_agent_cost": "0.54", + "total_run_cost": "58.29", + "average_steps": "10.82", + "percent_finished": "0.89" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2936,15 +2936,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2976,14 +2976,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.88, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.35", - "total_run_cost": "40.25", - "average_steps": "12.71", + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", "percent_finished": "1.0" } }, @@ -2991,8 +2991,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -3004,15 +3004,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -3044,14 +3044,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.6852, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.21", + "total_run_cost": "25.48", + "average_steps": "9.9", "percent_finished": "1.0" } }, @@ -3059,8 +3059,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -3072,8 +3072,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } diff --git a/data/models/google_gemini-3-pro.json b/data/models/google_gemini-3-pro.json index 11bc2c604594dba4ea7261f6d2e8fc51b386c982..227708133a9fa942407fc14b35830ec5101b4955 100644 --- a/data/models/google_gemini-3-pro.json +++ b/data/models/google_gemini-3-pro.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.1, + "score": 56.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-21", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.0, + "score": 56.9, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-06", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 69.4, + "score": 62.2, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,7 +339,7 @@ "max_score": 100.0 }, "score_details": { - "score": 61.8, + "score": 61.1, "uncertainty": { "standard_error": { "value": 2.8 @@ -349,7 +349,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-21", + "evaluation_timestamp": "2026-01-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.9, + "score": 69.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -454,7 +454,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -478,7 +478,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -487,17 +487,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.2, + "score": 61.8, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -514,7 +514,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3.1-pro.json b/data/models/google_gemini-3.1-pro.json index ecc327e910f8e4125f6002db506fbf114eb7832a..d5a56163d0a60a102dbdbef6cd39c4565d8d4e26 100644 --- a/data/models/google_gemini-3.1-pro.json +++ b/data/models/google_gemini-3.1-pro.json @@ -4,13 +4,13 @@ "id": "google/gemini-3.1-pro", "developer": "Google", "additional_details": { - "agent_name": "Forge Code", - "agent_organization": "Forge Code" + "agent_name": "Terminus-KIRA", + "agent_organization": "KRAFTON AI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-02", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 78.4, + "score": 74.8, "uncertainty": { "standard_error": { - "value": 1.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-03-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.8, + "score": 78.4, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 1.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini_3_pro.json b/data/models/google_gemini_3_pro.json index 104ce8e17e340df7c6fe6dc4bc1f7866f9c1cc71..8f4f634348fdc10ff492053d2a6241085b80ddd7 100644 --- a/data/models/google_gemini_3_pro.json +++ b/data/models/google_gemini_3_pro.json @@ -6,6 +6,78 @@ "inference_platform": "unknown" }, "evaluations": [ + { + "evaluation_id": "ace/google_gemini-3-pro/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Overall Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Overall ACE score (paper snapshot, approximate).", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.47 + }, + "generation_config": { + "additional_details": { + "run_setting": "High", + "value_quality": "approximate" + } + } + }, + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.509 + }, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High", + "value_quality": "approximate" + } + } + }, { "evaluation_id": "apex-agents/google_gemini-3-pro/1773260200", "retrieved_timestamp": "1773260200", @@ -205,78 +277,6 @@ } } }, - { - "evaluation_id": "ace/google_gemini-3-pro/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Overall Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Overall ACE score (paper snapshot, approximate).", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.47 - }, - "generation_config": { - "additional_details": { - "run_setting": "High", - "value_quality": "approximate" - } - } - }, - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.509 - }, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High", - "value_quality": "approximate" - } - } - }, { "evaluation_id": "apex-v1/google_gemini-3-pro/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/google_gemma-2-2b.json b/data/models/google_gemma-2-2b.json index 3875c2305ff98601b733940bab48bde956384743..853dbf9408f0d97b5b090784ef420b0479e50847 100644 --- a/data/models/google_gemma-2-2b.json +++ b/data/models/google_gemma-2-2b.json @@ -5,7 +5,7 @@ "developer": "Google", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "InternLM2ForCausalLM", "params_billions": "2.614" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2018 + "score": 0.1993 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3709 + "score": 0.3656 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0302 + "score": 0.0287 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4219 + "score": 0.4232 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2217 + "score": 0.218 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1993 + "score": 0.2018 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3656 + "score": 0.3709 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0287 + "score": 0.0302 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.4219 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.218 + "score": 0.2217 } } ], diff --git a/data/models/gunulhona_gemma-ko-merge-peft.json b/data/models/gunulhona_gemma-ko-merge-peft.json index 7aaca8e76ac4d54a166ddba85afcaa38d76748c4..632db743d31cc31d893a565c873ea7be7cc73fb2 100644 --- a/data/models/gunulhona_gemma-ko-merge-peft.json +++ b/data/models/gunulhona_gemma-ko-merge-peft.json @@ -5,7 +5,7 @@ "developer": "Gunulhona", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "?", "params_billions": "20.318" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4441 + "score": 0.288 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4863 + "score": 0.5154 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3247 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3986 + "score": 0.408 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3098 + "score": 0.3817 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.288 + "score": 0.4441 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5154 + "score": 0.4863 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.307 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.408 + "score": 0.3986 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3817 + "score": 0.3098 } } ], diff --git a/data/models/huggingfacetb_smollm2-360m-instruct.json b/data/models/huggingfacetb_smollm2-360m-instruct.json index b9a479d91c3b9226f2e42a0a5392df69dc01d9df..a4edda6b87f3cb7e0eeabaee9333bba91d019d12 100644 --- a/data/models/huggingfacetb_smollm2-360m-instruct.json +++ b/data/models/huggingfacetb_smollm2-360m-instruct.json @@ -5,9 +5,9 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", - "params_billions": "0.36" + "params_billions": "0.362" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.083 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3144 + "score": 0.3053 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0151 + "score": 0.0083 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.255 + "score": 0.2651 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3461 + "score": 0.3423 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1117 + "score": 0.1126 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.083 + "score": 0.3842 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3053 + "score": 0.3144 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0083 + "score": 0.0151 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2651 + "score": 0.255 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3423 + "score": 0.3461 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1117 } } ], diff --git a/data/models/internlm_internlm2-1_8b-reward.json b/data/models/internlm_internlm2-1_8b-reward.json index fdd95af043dae0ccf6943006b6d82bc9a69847ba..db02a96edd7d4deec66c9db68ed23c5ecfdc96f9 100644 --- a/data/models/internlm_internlm2-1_8b-reward.json +++ b/data/models/internlm_internlm2-1_8b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", + "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3902 + "score": 0.8217 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2758 + "score": 0.9358 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.6623 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4426 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4711 + "score": 0.8724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.596 + "score": 0.3902 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1934 + "score": 0.2758 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8217 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9358 + "score": 0.4426 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6623 + "score": 0.4711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8724 + "score": 0.1934 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json index 1160e36435338576bea29bef2eb39f295798e22d..88cf241ec28671fc887627a1e7bb1af09c15f9cf 100644 --- a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json +++ b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json @@ -5,7 +5,7 @@ "developer": "jaspionjader", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4345 + "score": 0.4418 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5419 + "score": 0.5406 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1292 + "score": 0.1352 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3062 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3854 + "score": 0.386 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4418 + "score": 0.4345 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.5419 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1352 + "score": 0.1292 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3087 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3854 } } ], diff --git a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json index 199a2574ee197c6be6eb6b38849e4c63ab43b085..90656456ac84eb015314278cafdaa288005eb1ba 100644 --- a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json +++ b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json @@ -5,7 +5,7 @@ "developer": "LeroyDyer", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "7.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3798 + "score": 0.3579 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4483 + "score": 0.4477 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.04 + "score": 0.0423 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3129 + "score": 0.3096 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4148 + "score": 0.4134 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2389 + "score": 0.2376 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3579 + "score": 0.3798 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4477 + "score": 0.4483 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0423 + "score": 0.04 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3129 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4134 + "score": 0.4148 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2376 + "score": 0.2389 } } ], diff --git a/data/models/llmat_mistral-v0.3-7b-orpo.json b/data/models/llmat_mistral-v0.3-7b-orpo.json index c2c9120f6bd010e2ff566effb35617377de554ee..3a1b947d84c5d76bb2423237a20cbf150415592e 100644 --- a/data/models/llmat_mistral-v0.3-7b-orpo.json +++ b/data/models/llmat_mistral-v0.3-7b-orpo.json @@ -5,7 +5,7 @@ "developer": "llmat", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "7.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.377 + "score": 0.364 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3978 + "score": 0.4005 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0242 + "score": 0.0015 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2668 + "score": 0.2693 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3555 + "score": 0.3529 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2278 + "score": 0.2301 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.364 + "score": 0.377 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4005 + "score": 0.3978 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0015 + "score": 0.0242 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2693 + "score": 0.2668 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3529 + "score": 0.3555 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2301 + "score": 0.2278 } } ], diff --git a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json index 018a3210e6da83ca7e6d3501fd6a7fefdb10449c..2ac3bb48cceaf59e0da754702b99c90edec98998 100644 --- a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json +++ b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json @@ -5,7 +5,7 @@ "developer": "Magpie-Align", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4118 + "score": 0.4027 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4811 + "score": 0.4789 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.034 + "score": 0.0461 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.2768 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3047 + "score": 0.3087 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3006 + "score": 0.3001 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4027 + "score": 0.4118 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.4811 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0461 + "score": 0.034 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2752 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3047 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3001 + "score": 0.3006 } } ], diff --git a/data/models/meta-llama_meta-llama-3-8b-instruct.json b/data/models/meta-llama_meta-llama-3-8b-instruct.json index 6b360a3e795c08b36f8a5ae66db62ef8475fd282..7f250b5cb54b80c03e9c7605296b8768c6176139 100644 --- a/data/models/meta-llama_meta-llama-3-8b-instruct.json +++ b/data/models/meta-llama_meta-llama-3-8b-instruct.json @@ -5,7 +5,7 @@ "developer": "meta-llama", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4782 + "score": 0.7408 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.491 + "score": 0.4989 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0914 + "score": 0.0869 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2928 + "score": 0.2592 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3568 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3591 + "score": 0.3664 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7408 + "score": 0.4782 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4989 + "score": 0.491 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0869 + "score": 0.0914 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2928 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3568 + "score": 0.3805 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3664 + "score": 0.3591 } } ], diff --git a/data/models/microsoft_phi-3-mini-4k-instruct.json b/data/models/microsoft_phi-3-mini-4k-instruct.json index 9787f21694f92686e734f02c35c722665943d7d4..f0214854e07bd3ad87162726609ad6d31e31396a 100644 --- a/data/models/microsoft_phi-3-mini-4k-instruct.json +++ b/data/models/microsoft_phi-3-mini-4k-instruct.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Phi3ForCausalLM", "params_billions": "3.821" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5613 + "score": 0.5477 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5676 + "score": 0.5491 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1163 + "score": 0.1639 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3196 + "score": 0.3322 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.395 + "score": 0.4284 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3866 + "score": 0.4022 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5477 + "score": 0.5613 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5491 + "score": 0.5676 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1639 + "score": 0.1163 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3322 + "score": 0.3196 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4284 + "score": 0.395 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4022 + "score": 0.3866 } } ], diff --git a/data/models/minimax_minimax-m2.1.json b/data/models/minimax_minimax-m2.1.json index 33905b0ffaa72b65b1d021a73faac651c1ebfecd..a5b957c9841bcbc5016fb37de174c988ce273877 100644 --- a/data/models/minimax_minimax-m2.1.json +++ b/data/models/minimax_minimax-m2.1.json @@ -4,13 +4,13 @@ "id": "minimax/minimax-m2.1", "developer": "MiniMax", "additional_details": { - "agent_name": "Crux", - "agent_organization": "Roam" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-22", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 36.6, + "score": 29.2, "uncertainty": { "standard_error": { "value": 2.9 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 36.6, "uncertainty": { "standard_error": { "value": 2.9 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/mistralai_mistral-medium-3.json b/data/models/mistralai_mistral-medium-3.json index 8c2cd1ffe5583f8e08ea6a503148c17defbde8f4..3d0755e0237b9760231fee380b173b29d57eab4a 100644 --- a/data/models/mistralai_mistral-medium-3.json +++ b/data/models/mistralai_mistral-medium-3.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/mistralai_mistral-small-2503.json b/data/models/mistralai_mistral-small-2503.json index 6df0d972b005ae32b060fdc4673d6092e770670f..be5d73de7278abb3c747dbebd44b60d3fa624503 100644 --- a/data/models/mistralai_mistral-small-2503.json +++ b/data/models/mistralai_mistral-small-2503.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/moonshot-ai_kimi-k2-instruct.json b/data/models/moonshot-ai_kimi-k2-instruct.json index 758984500ae56b028445a56fb8562c74a90a3a8f..2cfded0145e5c6821159f45b392f6b86e15c7f49 100644 --- a/data/models/moonshot-ai_kimi-k2-instruct.json +++ b/data/models/moonshot-ai_kimi-k2-instruct.json @@ -4,13 +4,13 @@ "id": "moonshot-ai/kimi-k2-instruct", "developer": "Moonshot AI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.7, + "score": 27.8, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.8, + "score": 26.7, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/multiple_multiple.json b/data/models/multiple_multiple.json index 45d5496f95937f9e2416eacde936df0b731a4892..fa0150e47673164e631d2fcaaf30ef2ab4f1a18b 100644 --- a/data/models/multiple_multiple.json +++ b/data/models/multiple_multiple.json @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-20", + "evaluation_timestamp": "2025-11-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,10 +43,10 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 50.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.7 }, "num_samples": 435 } @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-07", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.0, + "score": 58.4, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-11", + "evaluation_timestamp": "2025-11-20", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,10 +191,10 @@ "max_score": 100.0 }, "score_details": { - "score": 50.1, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.8 }, "num_samples": 435 } @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-03-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4, + "score": 71.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/nazimali_mistral-nemo-kurdish-instruct.json b/data/models/nazimali_mistral-nemo-kurdish-instruct.json index bf12d1ac4ce4431e1cc4657892f1de48dd5df10b..7bcf436e7a29205a95d8b228b8e52bcfb9264e7a 100644 --- a/data/models/nazimali_mistral-nemo-kurdish-instruct.json +++ b/data/models/nazimali_mistral-nemo-kurdish-instruct.json @@ -5,7 +5,7 @@ "developer": "nazimali", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "12.248" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.486 + "score": 0.4964 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4721 + "score": 0.4699 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0846 + "score": 0.0045 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2844 + "score": 0.2827 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4006 + "score": 0.3979 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3063 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4964 + "score": 0.486 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4699 + "score": 0.4721 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0045 + "score": 0.0846 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2844 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3979 + "score": 0.4006 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3063 + "score": 0.3087 } } ], diff --git a/data/models/nicolinho_qrm-gemma-2-27b.json b/data/models/nicolinho_qrm-gemma-2-27b.json index 98185886d3c230ddcd90456c69a5aeed49795fc5..1dea90f885df9d34139a9ef21e55b3dcce1a25fd 100644 --- a/data/models/nicolinho_qrm-gemma-2-27b.json +++ b/data/models/nicolinho_qrm-gemma-2-27b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.9444 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7853 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3719 + "score": 0.9013 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.927 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9578 + "score": 0.9826 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9535 + "score": 0.7667 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8321 + "score": 0.7853 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9444 + "score": 0.3719 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9013 + "score": 0.9578 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.927 + "score": 0.9535 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9826 + "score": 0.8321 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/omkar1102_code-yi.json b/data/models/omkar1102_code-yi.json index 420be452467271270dd617b6ac43e82dcaa608b1..c43a8e6f44d50964e1b475e97cca0076acc3fcc2 100644 --- a/data/models/omkar1102_code-yi.json +++ b/data/models/omkar1102_code-yi.json @@ -5,7 +5,7 @@ "developer": "Omkar1102", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "2.084" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2254 + "score": 0.2148 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.276 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2508 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3762 + "score": 0.3802 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1123 + "score": 0.1126 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2148 + "score": 0.2254 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.275 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2508 + "score": 0.2576 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3762 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1123 } } ], diff --git a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json index f86ab61b521749eb0f8fa030a3c29f5a32d327a1..486ba5e61208261c68f73d7d2bf88d78b1b36131 100644 --- a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json +++ b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json @@ -5,7 +5,7 @@ "developer": "ontocord", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.759" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1162 + "score": 0.1128 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3184 + "score": 0.3171 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0113 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2634 + "score": 0.2685 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3447 + "score": 0.346 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1124 + "score": 0.1129 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1162 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3171 + "score": 0.3184 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0113 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2685 + "score": 0.2634 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.3447 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1124 } } ], diff --git a/data/models/openai_gpt-4o-mini-2024-07-18.json b/data/models/openai_gpt-4o-mini-2024-07-18.json index f38f4e1ddedf08dfff3e3ceeb97aeebb3dcce913..1b3fb4c30102ee1f603e3640bb4c4b14c38b8cac 100644 --- a/data/models/openai_gpt-4o-mini-2024-07-18.json +++ b/data/models/openai_gpt-4o-mini-2024-07-18.json @@ -2124,10 +2124,10 @@ } }, { - "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2146,128 +2146,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8007 + "score": 0.5796 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.4105 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6075 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8081 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8374 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5796 + "score": 0.7414 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2276,111 +2252,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4105 + "score": 0.6962 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.8007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.6075 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7414 + "score": 0.8081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6962 + "score": 0.8374 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openai_gpt-5-2025-08-07.json b/data/models/openai_gpt-5-2025-08-07.json index 0853492fcc4bbd45f07185718454686350fe1be2..fdb97ce4111978e08938be252c1bd72645c11969 100644 --- a/data/models/openai_gpt-5-2025-08-07.json +++ b/data/models/openai_gpt-5-2025-08-07.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/openai_gpt-5-codex.json b/data/models/openai_gpt-5-codex.json index 21a2358b27ce78cb6bd659dfe0536dac012c1818..c82c5f95f908380f688e9ef7118281bf4bf2ade4 100644 --- a/data/models/openai_gpt-5-codex.json +++ b/data/models/openai_gpt-5-codex.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 44.3, + "score": 43.4, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.4, + "score": 44.3, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-mini.json b/data/models/openai_gpt-5-mini.json index a9c814071acbfdd08e7ec7cb2178b705e66b8da6..1a99c04bbe99b006a4925655196388ed4bd20247 100644 --- a/data/models/openai_gpt-5-mini.json +++ b/data/models/openai_gpt-5-mini.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 31.9, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 24.0, + "score": 31.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 24.0, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-nano.json b/data/models/openai_gpt-5-nano.json index 0d64dfdab152e442151bbf86365c72d39472c605..6e06e0851504fafe04f9fd1896b5ae58a6354f7a 100644 --- a/data/models/openai_gpt-5-nano.json +++ b/data/models/openai_gpt-5-nano.json @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.9, + "score": 7.0, "uncertainty": { "standard_error": { "value": 1.9 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.0, + "score": 7.9, "uncertainty": { "standard_error": { "value": 1.9 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.2-2025-12-11.json b/data/models/openai_gpt-5.2-2025-12-11.json index b1415af8e1fa1580e027129e48746b8a6ec316b5..3ac8b435319869c3f1714af61314295abb8f963c 100644 --- a/data/models/openai_gpt-5.2-2025-12-11.json +++ b/data/models/openai_gpt-5.2-2025-12-11.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2-2025-12-11", "developer": "OpenAI", "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.22, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.36", - "total_run_cost": "36.37", - "average_steps": "10.05", - "percent_finished": "1.0" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.071, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.55", + "total_run_cost": "55.03", + "average_steps": "51.59", + "percent_finished": "0.61" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -193,8 +193,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -246,23 +246,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.22, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.36", + "total_run_cost": "36.37", + "average_steps": "10.05", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.071, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "55.03", - "average_steps": "51.59", - "percent_finished": "0.61" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -342,8 +342,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -418,7 +418,7 @@ } }, { - "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,14 +450,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.43, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.43", - "total_run_cost": "43.11", - "average_steps": "8.97", + "average_agent_cost": "0.38", + "total_run_cost": "38.21", + "average_steps": "14.27", "percent_finished": "1.0" } }, @@ -465,8 +465,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.46, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.38", - "total_run_cost": "38.21", - "average_steps": "14.27", - "percent_finished": "1.0" + "average_agent_cost": "0.3", + "total_run_cost": "29.78", + "average_steps": "8.14", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -546,8 +546,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -622,7 +622,7 @@ } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -654,23 +654,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.46, + "score": 0.43, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "29.78", - "average_steps": "8.14", - "percent_finished": "0.99" + "average_agent_cost": "0.43", + "total_run_cost": "43.11", + "average_steps": "8.97", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -769,7 +769,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -801,14 +801,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.94", + "total_run_cost": "93.98", + "average_steps": "23.99", "percent_finished": "1.0" } }, @@ -816,8 +816,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -829,15 +829,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -869,14 +869,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.5253, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.94", - "total_run_cost": "93.98", - "average_steps": "23.99", + "average_agent_cost": "0.45", + "total_run_cost": "44.58", + "average_steps": "19.98", "percent_finished": "1.0" } }, @@ -884,8 +884,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -897,15 +897,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -937,14 +937,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5455, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.26", - "total_run_cost": "25.64", - "average_steps": "20.44", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -952,8 +952,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -965,15 +965,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1005,14 +1005,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5253, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.45", - "total_run_cost": "44.58", - "average_steps": "19.98", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -1020,8 +1020,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1033,15 +1033,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1073,14 +1073,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5455, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.26", + "total_run_cost": "25.64", + "average_steps": "20.44", "percent_finished": "1.0" } }, @@ -1088,8 +1088,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1101,15 +1101,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1141,14 +1141,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5, + "score": 0.48, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "5.77", - "average_steps": "11.4", + "average_agent_cost": "0.21", + "total_run_cost": "11.23", + "average_steps": "10.18", "percent_finished": "1.0" } }, @@ -1156,8 +1156,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1169,15 +1169,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1209,14 +1209,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.5, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.11", + "total_run_cost": "5.77", + "average_steps": "11.4", "percent_finished": "1.0" } }, @@ -1224,8 +1224,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1237,15 +1237,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1277,14 +1277,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.23", - "average_steps": "10.18", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1292,8 +1292,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1305,8 +1305,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1517,7 +1517,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1549,14 +1549,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", + "average_agent_cost": "0.25", + "total_run_cost": "26.27", + "average_steps": "11.08", "percent_finished": "1.0" } }, @@ -1564,8 +1564,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1577,8 +1577,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1653,7 +1653,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1685,14 +1685,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "26.27", - "average_steps": "11.08", + "average_agent_cost": "0.11", + "total_run_cost": "12.27", + "average_steps": "10.33", "percent_finished": "1.0" } }, @@ -1700,8 +1700,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1713,15 +1713,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1768,8 +1768,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1781,15 +1781,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1836,8 +1836,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1849,15 +1849,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1889,14 +1889,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.53, + "score": 0.71, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.15", - "total_run_cost": "18.88", - "average_steps": "9.92", + "average_agent_cost": "0.3", + "total_run_cost": "35.31", + "average_steps": "10.11", "percent_finished": "1.0" } }, @@ -1904,8 +1904,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1917,15 +1917,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1957,23 +1957,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.1", + "total_run_cost": "15.15", + "average_steps": "9.36", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1985,15 +1985,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2025,14 +2025,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.53, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "35.31", - "average_steps": "10.11", + "average_agent_cost": "0.15", + "total_run_cost": "18.88", + "average_steps": "9.92", "percent_finished": "1.0" } }, @@ -2040,8 +2040,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2053,15 +2053,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2093,23 +2093,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.1", - "total_run_cost": "15.15", - "average_steps": "9.36", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2121,8 +2121,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } diff --git a/data/models/openai_gpt-5.2.json b/data/models/openai_gpt-5.2.json index 77930b98006acfd2c6efa9a1e831cbd44874ee48..b67565fdc4a3c74776c9cdbc79f6dbe213fd591f 100644 --- a/data/models/openai_gpt-5.2.json +++ b/data/models/openai_gpt-5.2.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2", "developer": "OpenAI", "additional_details": { - "agent_name": "Codex CLI", - "agent_organization": "OpenAI" + "agent_name": "Mux", + "agent_organization": "Coder" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,11 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, - "uncertainty": { - "standard_error": { - "value": 3.0 - }, - "num_samples": 435 - } + "score": 60.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +64,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +185,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.0, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +212,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +226,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +250,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,11 +259,17 @@ "max_score": 100.0 }, "score_details": { - "score": 60.7 + "score": 54.0, + "uncertainty": { + "standard_error": { + "value": 2.9 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.3-codex.json b/data/models/openai_gpt-5.3-codex.json index 29bf24de6a839acf45f0dd14a04022b9052f2277..c4bd37007eca31e6c37beeb80d8affa5bcab3cb9 100644 --- a/data/models/openai_gpt-5.3-codex.json +++ b/data/models/openai_gpt-5.3-codex.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.3-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Simple Codex", + "agent_organization": "OpenAI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.7, + "score": 75.1, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 75.1, + "score": 77.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-24", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 77.3, + "score": 64.7, "uncertainty": { "standard_error": { - "value": 2.2 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.json b/data/models/openai_gpt-5.json index 9005bf4e58d9a5ca4b1df67ba70d59695c801ffb..55a3f1373a6ae761230df9b4daabc2f189cbdd8d 100644 --- a/data/models/openai_gpt-5.json +++ b/data/models/openai_gpt-5.json @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.2, + "score": 49.6, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 49.6, + "score": 35.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_o4-mini-2025-04-16.json b/data/models/openai_o4-mini-2025-04-16.json index 2445a24c2417bbf08441a2388a2fba31fa9d6f49..751e567f80107ae7403958c0915864809e731407 100644 --- a/data/models/openai_o4-mini-2025-04-16.json +++ b/data/models/openai_o4-mini-2025-04-16.json @@ -749,13 +749,13 @@ } }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", - "retrieved_timestamp": "1770683238.099205", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", + "retrieved_timestamp": "1760492095.8105888", "source_metadata": { + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation", - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party" + "source_type": "documentation" }, "eval_library": { "name": "unknown", @@ -765,62 +765,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", + "metric_config": { + "evaluation_description": "Pass@1 on Hard Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.014084507042253521 + }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Medium Problems", "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.0143 - } - }, - { - "evaluation_name": "Medium Problems", + "score": 0.30985915492957744 + }, "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Easy Problems", "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Easy Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.2923 - } - }, - { - "evaluation_name": "Easy Problems", + "score": 0.8873239436619719 + }, "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] - }, - "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.8571 } } ], @@ -828,13 +828,13 @@ "generation_config": null }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", - "retrieved_timestamp": "1760492095.8105888", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", + "retrieved_timestamp": "1770683238.099205", "source_metadata": { - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation" + "source_type": "documentation", + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party" }, "eval_library": { "name": "unknown", @@ -844,62 +844,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", - "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.014084507042253521 - }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Medium Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Hard Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.30985915492957744 - }, + "score": 0.0143 + } + }, + { + "evaluation_name": "Medium Problems", "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Easy Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.8873239436619719 - }, + "score": 0.2923 + } + }, + { + "evaluation_name": "Easy Problems", "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] + }, + "metric_config": { + "evaluation_description": "Pass@1 on Easy Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.8571 } } ], diff --git a/data/models/openbmb_eurus-rm-7b.json b/data/models/openbmb_eurus-rm-7b.json index e1154a660c89f219cec7ec844602d9d8fdfa08ad..44637ca17276f082725f61866dcfa3229cc54274 100644 --- a/data/models/openbmb_eurus-rm-7b.json +++ b/data/models/openbmb_eurus-rm-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5806 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6 + "score": 0.8159 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9804 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5683 + "score": 0.6557 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6267 + "score": 0.8135 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7475 + "score": 0.8633 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5972 + "score": 0.7172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8159 + "score": 0.5806 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9804 + "score": 0.6 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5683 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8135 + "score": 0.6267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8633 + "score": 0.7475 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7172 + "score": 0.5972 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openbmb_ultrarm-13b.json b/data/models/openbmb_ultrarm-13b.json index 84bdd483e26f91976f0250093c6ea14b4f0ff97c..c52a509adb327ccd9798d5a844c78601940ebb17 100644 --- a/data/models/openbmb_ultrarm-13b.json +++ b/data/models/openbmb_ultrarm-13b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6903 + "score": 0.4683 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.5063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5548 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5519 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5986 + "score": 0.5089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6244 + "score": 0.6081 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7294 + "score": 0.3036 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4683 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5063 + "score": 0.6903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5519 + "score": 0.5548 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5089 + "score": 0.5986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6081 + "score": 0.6244 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.7294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v1.0-cost.json b/data/models/pku-alignment_beaver-7b-v1.0-cost.json index 3777eba3edfdc470c669a503ac85994bf8139135..8e786484059e3101c1249c1cce8c4b2be81faa5a 100644 --- a/data/models/pku-alignment_beaver-7b-v1.0-cost.json +++ b/data/models/pku-alignment_beaver-7b-v1.0-cost.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3332 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.5798 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2313 + "score": 0.6173 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3989 + "score": 0.4232 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7589 + "score": 0.7351 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2939 + "score": 0.5482 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.57 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5798 + "score": 0.3332 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6173 + "score": 0.3263 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.2313 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3989 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7351 + "score": 0.7589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5482 + "score": 0.2939 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/primeintellect_intellect-1.json b/data/models/primeintellect_intellect-1.json index b8fa0710a43fde559062289de677b131f6527599..7d9ec915394b34f60309abc9ed073a5aa8bce5ab 100644 --- a/data/models/primeintellect_intellect-1.json +++ b/data/models/primeintellect_intellect-1.json @@ -5,7 +5,7 @@ "developer": "PrimeIntellect", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "10.211" } @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.274 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.25 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3339 + "score": 0.3753 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1123 + "score": 0.112 } } ], @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.274 + "score": 0.276 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.25 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3753 + "score": 0.3339 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.112 + "score": 0.1123 } } ], diff --git a/data/models/qingy2019_oracle-14b.json b/data/models/qingy2019_oracle-14b.json index 8fc1dd082923a347043a18d5c18fb5d4904fae42..22bb2fc596a8b7a91dfafd1db7c00009286e84b4 100644 --- a/data/models/qingy2019_oracle-14b.json +++ b/data/models/qingy2019_oracle-14b.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MixtralForCausalLM", "params_billions": "13.668" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2401 + "score": 0.2358 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4622 + "score": 0.4612 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0725 + "score": 0.0642 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2576 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3703 + "score": 0.3717 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2379 + "score": 0.2382 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2358 + "score": 0.2401 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4612 + "score": 0.4622 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0642 + "score": 0.0725 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2609 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3717 + "score": 0.3703 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2382 + "score": 0.2379 } } ], diff --git a/data/models/qingy2019_qwen2.5-math-14b-instruct.json b/data/models/qingy2019_qwen2.5-math-14b-instruct.json index 21a12461d0cce8974037575db54605a0da19b661..d07b0024c79975f453423bba6d277d902ff3056c 100644 --- a/data/models/qingy2019_qwen2.5-math-14b-instruct.json +++ b/data/models/qingy2019_qwen2.5-math-14b-instruct.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6005 + "score": 0.6066 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6356 + "score": 0.635 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2764 + "score": 0.3716 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3691 + "score": 0.3725 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5339 + "score": 0.5331 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.6005 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.635 + "score": 0.6356 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.2764 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3725 + "score": 0.3691 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5331 + "score": 0.5339 } } ], diff --git a/data/models/ray2333_grm-llama3-8b-distill.json b/data/models/ray2333_grm-llama3-8b-distill.json index a698e2dc0120e472c95dbd375c2ce72c243acf20..28bee941f1d39ce464ca73210ca8cbef67a5d082 100644 --- a/data/models/ray2333_grm-llama3-8b-distill.json +++ b/data/models/ray2333_grm-llama3-8b-distill.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8464 + "score": 0.589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5874 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6842 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5902 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9133 + "score": 0.6727 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7209 + "score": 0.5743 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-distill/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-distill/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.589 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5874 + "score": 0.8464 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5902 + "score": 0.6842 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7222 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6727 + "score": 0.9133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5743 + "score": 0.7209 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/replete-ai_replete-llm-qwen2-7b.json b/data/models/replete-ai_replete-llm-qwen2-7b.json index e51b10d1bb4ff7d76603042a5cdf25b16b9bc87f..627d67572ee0cd0135e173767115d5ee7360b5e6 100644 --- a/data/models/replete-ai_replete-llm-qwen2-7b.json +++ b/data/models/replete-ai_replete-llm-qwen2-7b.json @@ -5,7 +5,7 @@ "developer": "Replete-AI", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0932 + "score": 0.0905 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2985 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2475 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3941 + "score": 0.3848 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1157 + "score": 0.1158 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0905 + "score": 0.0932 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2985 + "score": 0.2977 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2475 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3848 + "score": 0.3941 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1158 + "score": 0.1157 } } ], diff --git a/data/models/riaz_finellama-3.1-8b.json b/data/models/riaz_finellama-3.1-8b.json index 554fa555a5c7288aadab0b51655132daffa7eed6..926e4908b22e1e63c1ff9b2a4076f85fbbd8288e 100644 --- a/data/models/riaz_finellama-3.1-8b.json +++ b/data/models/riaz_finellama-3.1-8b.json @@ -5,7 +5,7 @@ "developer": "riaz", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4373 + "score": 0.4137 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4586 + "score": 0.4565 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0514 + "score": 0.0453 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.276 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3763 + "score": 0.3776 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2964 + "score": 0.2978 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4137 + "score": 0.4373 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4565 + "score": 0.4586 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0453 + "score": 0.0514 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2752 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3776 + "score": 0.3763 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2978 + "score": 0.2964 } } ], diff --git a/data/models/sao10k_l3-70b-euryale-v2.1.json b/data/models/sao10k_l3-70b-euryale-v2.1.json index 66fd342cda2fcd43939fb102adcaae988f1eb857..c24d09a959df8d13e2602095ebcf1b7a3f8d330c 100644 --- a/data/models/sao10k_l3-70b-euryale-v2.1.json +++ b/data/models/sao10k_l3-70b-euryale-v2.1.json @@ -5,7 +5,7 @@ "developer": "Sao10K", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "70.554" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7281 + "score": 0.7384 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6503 + "score": 0.6471 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2243 + "score": 0.2137 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4196 + "score": 0.4209 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5096 + "score": 0.5104 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7384 + "score": 0.7281 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6471 + "score": 0.6503 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2137 + "score": 0.2243 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4209 + "score": 0.4196 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5104 + "score": 0.5096 } } ], diff --git a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json index ecef2ed14b1e2210796bfe9e4943404f7ebd3b7c..83ff30916d3f35c6333d5f652927c45d63836756 100644 --- a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json +++ b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6292 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5916 + "score": 0.8338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9944 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.6513 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7051 + "score": 0.8644 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6647 + "score": 0.7492 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8338 + "score": 0.6292 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9944 + "score": 0.5916 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6513 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8644 + "score": 0.7051 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7492 + "score": 0.6647 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/skywork_skywork-reward-gemma-2-27b.json b/data/models/skywork_skywork-reward-gemma-2-27b.json index c2b742dbc17aa9720f66d090e08eb784ec7accdc..0b4cc11a7f17a4535cc37f935fdae034a6214bce 100644 --- a/data/models/skywork_skywork-reward-gemma-2-27b.json +++ b/data/models/skywork_skywork-reward-gemma-2-27b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7576 + "score": 0.938 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7368 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4031 + "score": 0.9145 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7049 + "score": 0.9189 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9422 + "score": 0.9606 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9323 + "score": 0.7576 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8261 + "score": 0.7368 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.938 + "score": 0.4031 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.7049 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9145 + "score": 0.9422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9189 + "score": 0.9323 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9606 + "score": 0.8261 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/skywork_skywork-reward-llama-3.1-8b.json b/data/models/skywork_skywork-reward-llama-3.1-8b.json index dedd0015bc30c7a59cc8db5e1fbdb3b6b6cbc978..d700e3f49377a01e2f9d11b4773879e002a65e98 100644 --- a/data/models/skywork_skywork-reward-llama-3.1-8b.json +++ b/data/models/skywork_skywork-reward-llama-3.1-8b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9252 + "score": 0.7314 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.6989 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8728 + "score": 0.425 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9081 + "score": 0.6284 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.962 + "score": 0.9333 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7314 + "score": 0.9616 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6989 + "score": 0.741 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Llama-3.1-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.425 + "score": 0.9252 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9333 + "score": 0.8728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9616 + "score": 0.9081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.741 + "score": 0.962 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/skywork_skywork-vl-reward-7b.json b/data/models/skywork_skywork-vl-reward-7b.json index 651d1416fd84d9618565234fc2f23befa272cb51..d1caca7afd32adac0ce3eaccc5894c6b1d1db99d 100644 --- a/data/models/skywork_skywork-vl-reward-7b.json +++ b/data/models/skywork_skywork-vl-reward-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Skywork_Skywork-VL-Reward-7B/1766412838.146816", + "evaluation_id": "reward-bench/Skywork_Skywork-VL-Reward-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6885 + "score": 0.9007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6063 + "score": 0.8994 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.875 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6339 + "score": 0.9108 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8911 + "score": 0.9176 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Skywork_Skywork-VL-Reward-7B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8909 + "score": 0.6885 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7586 + "score": 0.6063 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Skywork_Skywork-VL-Reward-7B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9007 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8994 + "score": 0.6339 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.875 + "score": 0.8911 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9108 + "score": 0.8909 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9176 + "score": 0.7586 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json b/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json index a439bc62eb51d7fa9519bb2716501909fa712711..1f18d691fd936d47f7bbe51ac3217aef426c667e 100644 --- a/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json +++ b/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json @@ -5,7 +5,7 @@ "developer": "sometimesanotion", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5367 + "score": 0.5278 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6561 + "score": 0.6557 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.358 + "score": 0.3119 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3867 + "score": 0.3842 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.474 + "score": 0.4754 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5395 + "score": 0.5396 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5278 + "score": 0.5367 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.6561 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3119 + "score": 0.358 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.3867 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4754 + "score": 0.474 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5396 + "score": 0.5395 } } ], diff --git a/data/models/spow12_chatwaifu_v2.0_22b.json b/data/models/spow12_chatwaifu_v2.0_22b.json index ca25894b34b4dcd9cda648ad0e73ccd6dac87171..b7b1556be3e6af250728822417ad3ec7fa2fe00c 100644 --- a/data/models/spow12_chatwaifu_v2.0_22b.json +++ b/data/models/spow12_chatwaifu_v2.0_22b.json @@ -5,7 +5,7 @@ "developer": "spow12", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "22.247" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6517 + "score": 0.6511 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5908 + "score": 0.5926 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2032 + "score": 0.1858 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3247 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3812 + "score": 0.3836 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6511 + "score": 0.6517 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5926 + "score": 0.5908 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1858 + "score": 0.2032 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.3238 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3836 + "score": 0.3812 } } ], diff --git a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json index 5d50cd0f56f0ac33470efc8c81f51c02a702598e..7fb384620bce09b34b0d31ff763d2b1e971d0558 100644 --- a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json +++ b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json @@ -5,7 +5,7 @@ "developer": "tanliboy", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "9.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4501 + "score": 0.1829 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5472 + "score": 0.5488 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0944 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3104 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4017 + "score": 0.4056 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3792 + "score": 0.3805 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1829 + "score": 0.4501 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5488 + "score": 0.5472 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0944 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.3138 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4056 + "score": 0.4017 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3792 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-cobalt.json b/data/models/valiantlabs_llama3.1-8b-cobalt.json index c47cbaf67f9d28f7ee6c973b80a6e624aec4a0cc..e1d1a85c95f8deab607165ea40b9f93f0dba9c6d 100644 --- a/data/models/valiantlabs_llama3.1-8b-cobalt.json +++ b/data/models/valiantlabs_llama3.1-8b-cobalt.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3496 + "score": 0.7168 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4947 + "score": 0.4911 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1269 + "score": 0.1533 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3037 + "score": 0.2861 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3512 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3644 + "score": 0.3663 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7168 + "score": 0.3496 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4911 + "score": 0.4947 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1533 + "score": 0.1269 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3037 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3512 + "score": 0.3959 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3663 + "score": 0.3644 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json index f3e37b204fa779a9e21a0521a813b464c3fe641b..0736460b872bea97a209be68cef1f113fa7d9f3d 100644 --- a/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json +++ b/data/models/valiantlabs_llama3.1-8b-shiningvaliant2.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6496 + "score": 0.2678 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4774 + "score": 0.4429 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0566 + "score": 0.0521 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.302 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3909 + "score": 0.3959 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3382 + "score": 0.2927 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2678 + "score": 0.6496 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4429 + "score": 0.4774 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0566 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.302 + "score": 0.3104 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3909 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2927 + "score": 0.3382 } } ], diff --git a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json index 564c2ceeb0768d947ec7e8507c351558f76e3907..f9d2b14f0a0f79f11e39957c0f38b89aa1e78ac9 100644 --- a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json +++ b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2498 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3642 + "score": 0.5027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3497 + "score": 0.3728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.24 + "score": 0.4149 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2384 + "score": 0.3281 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0315 + "score": 0.6564 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.2498 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.3642 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.275 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3497 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.24 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3281 + "score": 0.2384 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6564 + "score": 0.0315 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/weqweasdas_rm-gemma-2b.json b/data/models/weqweasdas_rm-gemma-2b.json index 23bdb9776ed416ae7ab56cbae8b9e5655ba27e0c..b1151d30882a530616e4e5f2252a198e5a25a8a2 100644 --- a/data/models/weqweasdas_rm-gemma-2b.json +++ b/data/models/weqweasdas_rm-gemma-2b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-2B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-2B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6549 + "score": 0.3057 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.3705 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4079 + "score": 0.2812 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.4317 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4986 + "score": 0.3311 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7637 + "score": 0.2343 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6652 + "score": 0.1851 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-2B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-2B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3057 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3705 + "score": 0.6549 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.9441 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4317 + "score": 0.4079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3311 + "score": 0.4986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2343 + "score": 0.7637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1851 + "score": 0.6652 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/weqweasdas_rm-mistral-7b.json b/data/models/weqweasdas_rm-mistral-7b.json index 014b8589e308fa2571a3ca85971364da616341ae..2c0c95b4657b4530753b94c6c05b68b220f49072 100644 --- a/data/models/weqweasdas_rm-mistral-7b.json +++ b/data/models/weqweasdas_rm-mistral-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7982 + "score": 0.596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.5937 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6053 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5956 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8703 + "score": 0.6911 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7736 + "score": 0.7293 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.753 + "score": 0.6226 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.596 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5937 + "score": 0.7982 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5956 + "score": 0.6053 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6911 + "score": 0.8703 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.7736 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6226 + "score": 0.753 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/xai_grok-3-mini.json b/data/models/xai_grok-3-mini.json index 6f7e913322e4d12afd1b4e9815b3c829b5eb051d..7fda2d0643a4e1f5eb9d98f1dc44e9ec85970d5d 100644 --- a/data/models/xai_grok-3-mini.json +++ b/data/models/xai_grok-3-mini.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/xai_grok-3-mini/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/xai_grok-4.json b/data/models/xai_grok-4.json index 2396de921bef3d760583d06f283159a7a4b7cc87..1a7ae0a5e37faeab9841b8816b2f9d9332f7d639 100644 --- a/data/models/xai_grok-4.json +++ b/data/models/xai_grok-4.json @@ -4,13 +4,13 @@ "id": "xai/grok-4", "developer": "xAI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.1, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 23.1, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/xai_grok-code-fast-1.json b/data/models/xai_grok-code-fast-1.json index 3dead5de67b9b087729b3bc778973bf9f0bdd596..80a497dcbe0e067252e23ff5d2567dcecf97e827 100644 --- a/data/models/xai_grok-code-fast-1.json +++ b/data/models/xai_grok-code-fast-1.json @@ -4,13 +4,13 @@ "id": "xai/grok-code-fast-1", "developer": "xAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 14.2, + "score": 25.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.8, + "score": 14.2, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/ycros_bagelmisterytour-v2-8x7b.json b/data/models/ycros_bagelmisterytour-v2-8x7b.json index c7b7f840ab350df665db2f8b289c03a4556651c3..ba69aabd10b1f09ccc48e0969d876027b03e3a4b 100644 --- a/data/models/ycros_bagelmisterytour-v2-8x7b.json +++ b/data/models/ycros_bagelmisterytour-v2-8x7b.json @@ -5,7 +5,7 @@ "developer": "ycros", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MixtralForCausalLM", "params_billions": "46.703" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5994 + "score": 0.6262 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5159 + "score": 0.5142 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0785 + "score": 0.0937 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3045 + "score": 0.3079 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4203 + "score": 0.4138 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3473 + "score": 0.3481 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6262 + "score": 0.5994 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5142 + "score": 0.5159 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.0785 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3079 + "score": 0.3045 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4138 + "score": 0.4203 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3481 + "score": 0.3473 } } ], diff --git a/data/models/yoyo-ai_qwen2.5-14b-yoyo-1010.json b/data/models/yoyo-ai_qwen2.5-14b-yoyo-1010.json index 2a296ce898176000fe4f075772b9bc2bcb8d3b71..e42112fd11a9b36734c8e8c4176443d8815fdc44 100644 --- a/data/models/yoyo-ai_qwen2.5-14b-yoyo-1010.json +++ b/data/models/yoyo-ai_qwen2.5-14b-yoyo-1010.json @@ -5,7 +5,7 @@ "developer": "YOYO-AI", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.77" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5899 + "score": 0.7905 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.654 + "score": 0.6406 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4509 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3834 + "score": 0.3163 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4744 + "score": 0.4181 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5376 + "score": 0.4944 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7905 + "score": 0.5899 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6406 + "score": 0.654 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.4509 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3163 + "score": 0.3834 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4181 + "score": 0.4744 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4944 + "score": 0.5376 } } ],