diff --git a/data/benchmarks.json b/data/benchmarks.json index 14d2e01a923545e2a82d10a5a7dbef2727e8e639..91a65d08f78f049bdab371c0c04a87419e88aa4a 100644 --- a/data/benchmarks.json +++ b/data/benchmarks.json @@ -47,6 +47,10 @@ "benchmark": "hfopenllm_v2", "model_count": 4496 }, + { + "benchmark": "la_leaderboard", + "model_count": 5 + }, { "benchmark": "livecodebenchpro", "model_count": 27 diff --git a/data/benchmarks/appworld_test_normal.json b/data/benchmarks/appworld_test_normal.json index f149d2e83dd4295170a7dd85ee296bc311ef3b12..08e6e07b896354a14f6c03de279c5f059fb1809d 100644 --- a/data/benchmarks/appworld_test_normal.json +++ b/data/benchmarks/appworld_test_normal.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "appworld/test_normal": 0.7 + "appworld/test_normal": 0.68 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "appworld/test_normal": 0.36 + "appworld/test_normal": 0.505 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "appworld/test_normal": 0.0 + "appworld/test_normal": 0.071 } } ] diff --git a/data/benchmarks/browsecompplus.json b/data/benchmarks/browsecompplus.json index 7beeaa9ca1194eca39b656d11dd6cb8fd5ccd590..eec2c0760e9494a2f30d11881f89d05119df34ad 100644 --- a/data/benchmarks/browsecompplus.json +++ b/data/benchmarks/browsecompplus.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "browsecompplus": 0.61 + "browsecompplus": 0.49 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "browsecompplus": 0.57 + "browsecompplus": 0.51 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "browsecompplus": 0.46 + "browsecompplus": 0.43 } } ] diff --git a/data/benchmarks/global-mmlu-lite.json b/data/benchmarks/global-mmlu-lite.json index d435988274520e6c53f552dcf3f5b6760c0ea503..6a0c6630713b5c17f51b01fc50869d775287a68a 100644 --- a/data/benchmarks/global-mmlu-lite.json +++ b/data/benchmarks/global-mmlu-lite.json @@ -315,7 +315,7 @@ { "model_id": "google/gemini-3-pro-preview", "name": "gemini-3-pro-preview", - "developer": "Google", + "developer": "google", "scores": { "Global MMLU Lite": 0.9453, "Culturally Sensitive": 0.9397, diff --git a/data/benchmarks/hfopenllm_v2.json b/data/benchmarks/hfopenllm_v2.json index 14cadd1c78c83808c8c6786afa0b5f2a1daac5c9..3b4661a1c6824a42a734f0a1e7e5e9d2ae7e9a10 100644 --- a/data/benchmarks/hfopenllm_v2.json +++ b/data/benchmarks/hfopenllm_v2.json @@ -1019,12 +1019,12 @@ "name": "Qwen2.5-1.5B-continuous-learnt", "developer": "AtAndDev", "scores": { - "IFEval": 0.4605, - "BBH": 0.4258, - "MATH Level 5": 0.0748, - "GPQA": 0.2659, - "MUSR": 0.3636, - "MMLU-PRO": 0.2812 + "IFEval": 0.4511, + "BBH": 0.4275, + "MATH Level 5": 0.1473, + "GPQA": 0.2701, + "MUSR": 0.3623, + "MMLU-PRO": 0.2806 } }, { @@ -3047,12 +3047,12 @@ "name": "AetherTOT", "developer": "Daemontatox", "scores": { - "IFEval": 0.4398, - "BBH": 0.5066, - "MATH Level 5": 0.1488, + "IFEval": 0.4383, + "BBH": 0.5034, + "MATH Level 5": 0.1443, "GPQA": 0.3238, - "MUSR": 0.4079, - "MMLU-PRO": 0.3804 + "MUSR": 0.4052, + "MMLU-PRO": 0.3778 } }, { @@ -3229,12 +3229,12 @@ "name": "PathfinderAI", "developer": "Daemontatox", "scores": { - "IFEval": 0.4855, - "BBH": 0.6627, - "MATH Level 5": 0.4841, - "GPQA": 0.3096, - "MUSR": 0.4256, - "MMLU-PRO": 0.5542 + "IFEval": 0.3745, + "BBH": 0.6668, + "MATH Level 5": 0.4758, + "GPQA": 0.3943, + "MUSR": 0.4858, + "MMLU-PRO": 0.5593 } }, { @@ -4009,12 +4009,12 @@ "name": "Llama-3.2-1B-SPIN-iter0", "developer": "DavieLion", "scores": { - "IFEval": 0.1549, - "BBH": 0.2937, - "MATH Level 5": 0.006, - "GPQA": 0.2576, + "IFEval": 0.1507, + "BBH": 0.293, + "MATH Level 5": 0.0, + "GPQA": 0.2534, "MUSR": 0.3565, - "MMLU-PRO": 0.1128 + "MMLU-PRO": 0.1125 } }, { @@ -8468,12 +8468,12 @@ "name": "Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.3417, - "BBH": 0.3292, - "MATH Level 5": 0.0023, - "GPQA": 0.2576, - "MUSR": 0.3249, - "MMLU-PRO": 0.1638 + "IFEval": 0.3472, + "BBH": 0.3268, + "MATH Level 5": 0.0891, + "GPQA": 0.2517, + "MUSR": 0.3262, + "MMLU-PRO": 0.1641 } }, { @@ -8585,12 +8585,12 @@ "name": "josie-7b-v6.0-step2000", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.7628, - "BBH": 0.5098, - "MATH Level 5": 0.0, - "GPQA": 0.2802, - "MUSR": 0.4579, - "MMLU-PRO": 0.4033 + "IFEval": 0.7598, + "BBH": 0.5107, + "MATH Level 5": 0.4237, + "GPQA": 0.2768, + "MUSR": 0.4539, + "MMLU-PRO": 0.4012 } }, { @@ -8741,12 +8741,12 @@ "name": "Gemma-Ko-Merge-PEFT", "developer": "Gunulhona", "scores": { - "IFEval": 0.4441, - "BBH": 0.4863, + "IFEval": 0.288, + "BBH": 0.5154, "MATH Level 5": 0.0, - "GPQA": 0.307, - "MUSR": 0.3986, - "MMLU-PRO": 0.3098 + "GPQA": 0.3247, + "MUSR": 0.408, + "MMLU-PRO": 0.3817 } }, { @@ -9157,12 +9157,12 @@ "name": "SmolLM2-135M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.2883, - "BBH": 0.3124, - "MATH Level 5": 0.003, - "GPQA": 0.2357, - "MUSR": 0.3662, - "MMLU-PRO": 0.1115 + "IFEval": 0.0593, + "BBH": 0.3135, + "MATH Level 5": 0.0144, + "GPQA": 0.2341, + "MUSR": 0.3871, + "MMLU-PRO": 0.1092 } }, { @@ -9183,12 +9183,12 @@ "name": "SmolLM2-360M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.3842, - "BBH": 0.3144, - "MATH Level 5": 0.0151, - "GPQA": 0.255, - "MUSR": 0.3461, - "MMLU-PRO": 0.1117 + "IFEval": 0.083, + "BBH": 0.3053, + "MATH Level 5": 0.0083, + "GPQA": 0.2651, + "MUSR": 0.3423, + "MMLU-PRO": 0.1126 } }, { @@ -9391,12 +9391,12 @@ "name": "JOSIEv4o-8b-stage1-v4", "developer": "Isaak-Carter", "scores": { - "IFEval": 0.2477, - "BBH": 0.4758, - "MATH Level 5": 0.0453, - "GPQA": 0.2911, - "MUSR": 0.3641, - "MMLU-PRO": 0.3292 + "IFEval": 0.2553, + "BBH": 0.4725, + "MATH Level 5": 0.0529, + "GPQA": 0.2919, + "MUSR": 0.3654, + "MMLU-PRO": 0.3316 } }, { @@ -14318,12 +14318,12 @@ "name": "Llama-3-8B-Magpie-Align-v0.1", "developer": "Magpie-Align", "scores": { - "IFEval": 0.4027, - "BBH": 0.4789, - "MATH Level 5": 0.0461, - "GPQA": 0.2768, - "MUSR": 0.3087, - "MMLU-PRO": 0.3001 + "IFEval": 0.4118, + "BBH": 0.4811, + "MATH Level 5": 0.034, + "GPQA": 0.2752, + "MUSR": 0.3047, + "MMLU-PRO": 0.3006 } }, { @@ -17217,12 +17217,12 @@ "name": "code-yi", "developer": "Omkar1102", "scores": { - "IFEval": 0.2148, - "BBH": 0.276, + "IFEval": 0.2254, + "BBH": 0.275, "MATH Level 5": 0.0, - "GPQA": 0.2508, - "MUSR": 0.3802, - "MMLU-PRO": 0.1126 + "GPQA": 0.2576, + "MUSR": 0.3762, + "MMLU-PRO": 0.1123 } }, { @@ -18283,12 +18283,12 @@ "name": "Casa-14b-sce", "developer": "Quazim0t0", "scores": { - "IFEval": 0.6654, - "BBH": 0.6901, - "MATH Level 5": 0.4698, - "GPQA": 0.3331, - "MUSR": 0.431, - "MMLU-PRO": 0.5426 + "IFEval": 0.6718, + "BBH": 0.6891, + "MATH Level 5": 0.4985, + "GPQA": 0.3339, + "MUSR": 0.4323, + "MMLU-PRO": 0.5408 } }, { @@ -19557,12 +19557,12 @@ "name": "Qwen2.5-0.5B-Instruct", "developer": "Qwen", "scores": { - "IFEval": 0.3153, - "BBH": 0.3322, - "MATH Level 5": 0.1035, - "GPQA": 0.2592, - "MUSR": 0.3342, - "MMLU-PRO": 0.172 + "IFEval": 0.3071, + "BBH": 0.3341, + "MATH Level 5": 0.0, + "GPQA": 0.2576, + "MUSR": 0.3329, + "MMLU-PRO": 0.1697 } }, { @@ -19817,12 +19817,12 @@ "name": "Qwen2.5-Coder-7B-Instruct", "developer": "Qwen", "scores": { - "IFEval": 0.6147, - "BBH": 0.4999, - "MATH Level 5": 0.031, - "GPQA": 0.2936, - "MUSR": 0.4099, - "MMLU-PRO": 0.3354 + "IFEval": 0.6101, + "BBH": 0.5008, + "MATH Level 5": 0.3716, + "GPQA": 0.2919, + "MUSR": 0.4073, + "MMLU-PRO": 0.3352 } }, { @@ -20077,12 +20077,12 @@ "name": "Replete-LLM-Qwen2-7b", "developer": "Replete-AI", "scores": { - "IFEval": 0.0932, - "BBH": 0.2977, + "IFEval": 0.0905, + "BBH": 0.2985, "MATH Level 5": 0.0, - "GPQA": 0.2475, - "MUSR": 0.3941, - "MMLU-PRO": 0.1157 + "GPQA": 0.2534, + "MUSR": 0.3848, + "MMLU-PRO": 0.1158 } }, { @@ -24744,12 +24744,12 @@ "name": "Llama-3-Instruct-8B-SPPO-Iter3", "developer": "UCLA-AGI", "scores": { - "IFEval": 0.6703, - "BBH": 0.5076, - "MATH Level 5": 0.0718, + "IFEval": 0.6834, + "BBH": 0.508, + "MATH Level 5": 0.0959, "GPQA": 0.2651, - "MUSR": 0.3647, - "MMLU-PRO": 0.3658 + "MUSR": 0.3661, + "MMLU-PRO": 0.3644 } }, { @@ -25212,12 +25212,12 @@ "name": "Llama3.1-8B-ShiningValiant2", "developer": "ValiantLabs", "scores": { - "IFEval": 0.2678, - "BBH": 0.4429, - "MATH Level 5": 0.0521, - "GPQA": 0.302, - "MUSR": 0.3959, - "MMLU-PRO": 0.2927 + "IFEval": 0.6496, + "BBH": 0.4774, + "MATH Level 5": 0.0566, + "GPQA": 0.3104, + "MUSR": 0.3909, + "MMLU-PRO": 0.3382 } }, { @@ -26603,12 +26603,12 @@ "name": "autotrain-0tmgq-5tpbg", "developer": "abhishek", "scores": { - "IFEval": 0.1957, - "BBH": 0.3135, - "MATH Level 5": 0.0, - "GPQA": 0.2517, - "MUSR": 0.365, - "MMLU-PRO": 0.1151 + "IFEval": 0.1952, + "BBH": 0.3127, + "MATH Level 5": 0.0128, + "GPQA": 0.2592, + "MUSR": 0.3584, + "MMLU-PRO": 0.1144 } }, { @@ -27006,12 +27006,12 @@ "name": "Llama-3.1-Tulu-3-70B", "developer": "allenai", "scores": { - "IFEval": 0.8379, - "BBH": 0.6157, - "MATH Level 5": 0.3829, + "IFEval": 0.8291, + "BBH": 0.6164, + "MATH Level 5": 0.4502, "GPQA": 0.3733, - "MUSR": 0.4988, - "MMLU-PRO": 0.4656 + "MUSR": 0.4948, + "MMLU-PRO": 0.4645 } }, { @@ -27045,12 +27045,12 @@ "name": "Llama-3.1-Tulu-3-8B", "developer": "allenai", "scores": { - "IFEval": 0.8267, - "BBH": 0.405, - "MATH Level 5": 0.1964, - "GPQA": 0.2987, + "IFEval": 0.8255, + "BBH": 0.4061, + "MATH Level 5": 0.2115, + "GPQA": 0.297, "MUSR": 0.4175, - "MMLU-PRO": 0.2827 + "MMLU-PRO": 0.2821 } }, { @@ -30451,12 +30451,12 @@ "name": "Llama-3.2-3B-Deep-Test", "developer": "bunnycore", "scores": { - "IFEval": 0.1775, - "BBH": 0.295, - "MATH Level 5": 0.0, - "GPQA": 0.2517, - "MUSR": 0.3647, - "MMLU-PRO": 0.1049 + "IFEval": 0.4652, + "BBH": 0.4531, + "MATH Level 5": 0.1284, + "GPQA": 0.2643, + "MUSR": 0.3394, + "MMLU-PRO": 0.3152 } }, { @@ -31738,12 +31738,12 @@ "name": "dolphin-2.9.2-Phi-3-Medium-abliterated", "developer": "cognitivecomputations", "scores": { - "IFEval": 0.3613, - "BBH": 0.6123, - "MATH Level 5": 0.1239, - "GPQA": 0.328, - "MUSR": 0.4112, - "MMLU-PRO": 0.4494 + "IFEval": 0.4124, + "BBH": 0.6383, + "MATH Level 5": 0.182, + "GPQA": 0.3289, + "MUSR": 0.4349, + "MMLU-PRO": 0.4525 } }, { @@ -31881,12 +31881,12 @@ "name": "llama-43m-beta", "developer": "cpayne1303", "scores": { - "IFEval": 0.1949, - "BBH": 0.2965, - "MATH Level 5": 0.0045, + "IFEval": 0.1916, + "BBH": 0.2977, + "MATH Level 5": 0.0, "GPQA": 0.2685, - "MUSR": 0.3885, - "MMLU-PRO": 0.1111 + "MUSR": 0.3872, + "MMLU-PRO": 0.1132 } }, { @@ -33597,12 +33597,12 @@ "name": "TheBeagle-v2beta-32B-MGS", "developer": "fblgit", "scores": { - "IFEval": 0.5181, - "BBH": 0.7033, - "MATH Level 5": 0.4947, - "GPQA": 0.3826, - "MUSR": 0.5008, - "MMLU-PRO": 0.5915 + "IFEval": 0.4503, + "BBH": 0.7035, + "MATH Level 5": 0.3943, + "GPQA": 0.401, + "MUSR": 0.5021, + "MMLU-PRO": 0.5911 } }, { @@ -34663,12 +34663,12 @@ "name": "flan-t5-xl", "developer": "google", "scores": { - "IFEval": 0.2237, - "BBH": 0.4531, - "MATH Level 5": 0.0076, - "GPQA": 0.2525, - "MUSR": 0.4181, - "MMLU-PRO": 0.2147 + "IFEval": 0.2207, + "BBH": 0.4537, + "MATH Level 5": 0.0008, + "GPQA": 0.2458, + "MUSR": 0.422, + "MMLU-PRO": 0.2142 } }, { @@ -37796,12 +37796,12 @@ "name": "Kosmos-EVAA-Fusion-8B", "developer": "jaspionjader", "scores": { - "IFEval": 0.4418, - "BBH": 0.5406, - "MATH Level 5": 0.1352, - "GPQA": 0.3062, + "IFEval": 0.4345, + "BBH": 0.5419, + "MATH Level 5": 0.1292, + "GPQA": 0.3087, "MUSR": 0.4277, - "MMLU-PRO": 0.386 + "MMLU-PRO": 0.3854 } }, { @@ -44205,12 +44205,12 @@ "name": "phi-4", "developer": "microsoft", "scores": { - "IFEval": 0.0488, - "BBH": 0.6703, - "MATH Level 5": 0.2787, - "GPQA": 0.401, + "IFEval": 0.0585, + "BBH": 0.6691, + "MATH Level 5": 0.3165, + "GPQA": 0.406, "MUSR": 0.5034, - "MMLU-PRO": 0.5295 + "MMLU-PRO": 0.5287 } }, { @@ -44309,12 +44309,12 @@ "name": "Trinity-2-Codestral-22B-v0.2", "developer": "migtissera", "scores": { - "IFEval": 0.4345, - "BBH": 0.5686, - "MATH Level 5": 0.0838, - "GPQA": 0.3003, - "MUSR": 0.4045, - "MMLU-PRO": 0.334 + "IFEval": 0.443, + "BBH": 0.5706, + "MATH Level 5": 0.0869, + "GPQA": 0.3079, + "MUSR": 0.4031, + "MMLU-PRO": 0.3354 } }, { @@ -44595,12 +44595,12 @@ "name": "Mixtral-8x7B-v0.1", "developer": "mistralai", "scores": { - "IFEval": 0.2415, - "BBH": 0.5087, - "MATH Level 5": 0.102, - "GPQA": 0.3138, - "MUSR": 0.4321, - "MMLU-PRO": 0.385 + "IFEval": 0.2326, + "BBH": 0.5098, + "MATH Level 5": 0.0937, + "GPQA": 0.3205, + "MUSR": 0.4413, + "MMLU-PRO": 0.3871 } }, { @@ -44829,12 +44829,12 @@ "name": "NeuralDaredevil-8B-abliterated", "developer": "mlabonne", "scores": { - "IFEval": 0.7561, - "BBH": 0.5111, - "MATH Level 5": 0.0906, - "GPQA": 0.3062, - "MUSR": 0.4019, - "MMLU-PRO": 0.3841 + "IFEval": 0.4162, + "BBH": 0.5124, + "MATH Level 5": 0.0853, + "GPQA": 0.3029, + "MUSR": 0.415, + "MMLU-PRO": 0.3802 } }, { @@ -46870,12 +46870,12 @@ "name": "franqwenstein-35b", "developer": "nisten", "scores": { - "IFEval": 0.3799, - "BBH": 0.6647, - "MATH Level 5": 0.3406, - "GPQA": 0.4035, - "MUSR": 0.494, - "MMLU-PRO": 0.5731 + "IFEval": 0.3914, + "BBH": 0.6591, + "MATH Level 5": 0.3044, + "GPQA": 0.3591, + "MUSR": 0.4681, + "MMLU-PRO": 0.5611 } }, { @@ -47702,12 +47702,12 @@ "name": "wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue", "developer": "ontocord", "scores": { - "IFEval": 0.1128, - "BBH": 0.3171, - "MATH Level 5": 0.0113, - "GPQA": 0.2685, - "MUSR": 0.346, - "MMLU-PRO": 0.1129 + "IFEval": 0.1162, + "BBH": 0.3184, + "MATH Level 5": 0.0076, + "GPQA": 0.2634, + "MUSR": 0.3447, + "MMLU-PRO": 0.1124 } }, { @@ -47884,12 +47884,12 @@ "name": "Llama-FinSent-S", "developer": "oopere", "scores": { - "IFEval": 0.2119, - "BBH": 0.3156, - "MATH Level 5": 0.0181, - "GPQA": 0.2567, + "IFEval": 0.2164, + "BBH": 0.3169, + "MATH Level 5": 0.0128, + "GPQA": 0.2584, "MUSR": 0.3832, - "MMLU-PRO": 0.113 + "MMLU-PRO": 0.1134 } }, { @@ -50952,12 +50952,12 @@ "name": "Oracle-14B", "developer": "qingy2019", "scores": { - "IFEval": 0.2358, - "BBH": 0.4612, - "MATH Level 5": 0.0642, - "GPQA": 0.2576, - "MUSR": 0.3717, - "MMLU-PRO": 0.2382 + "IFEval": 0.2401, + "BBH": 0.4622, + "MATH Level 5": 0.0725, + "GPQA": 0.2609, + "MUSR": 0.3703, + "MMLU-PRO": 0.2379 } }, { @@ -50965,12 +50965,12 @@ "name": "Qwen2.5-Math-14B-Instruct", "developer": "qingy2019", "scores": { - "IFEval": 0.6066, - "BBH": 0.635, - "MATH Level 5": 0.3716, - "GPQA": 0.3725, + "IFEval": 0.6005, + "BBH": 0.6356, + "MATH Level 5": 0.2764, + "GPQA": 0.3691, "MUSR": 0.4757, - "MMLU-PRO": 0.5331 + "MMLU-PRO": 0.5339 } }, { @@ -51329,12 +51329,12 @@ "name": "Gemma-2-Ataraxy-Gemmasutra-9B-slerp", "developer": "recoilme", "scores": { - "IFEval": 0.7649, - "BBH": 0.5974, - "MATH Level 5": 0.0174, - "GPQA": 0.3305, - "MUSR": 0.4245, - "MMLU-PRO": 0.4207 + "IFEval": 0.2854, + "BBH": 0.5984, + "MATH Level 5": 0.1005, + "GPQA": 0.3297, + "MUSR": 0.4607, + "MMLU-PRO": 0.4162 } }, { @@ -51628,12 +51628,12 @@ "name": "Rombos-LLM-V2.5.1-Qwen-3b", "developer": "rombodawg", "scores": { - "IFEval": 0.2595, - "BBH": 0.3884, - "MATH Level 5": 0.0914, - "GPQA": 0.2743, + "IFEval": 0.2566, + "BBH": 0.39, + "MATH Level 5": 0.1208, + "GPQA": 0.2626, "MUSR": 0.3991, - "MMLU-PRO": 0.2719 + "MMLU-PRO": 0.2741 } }, { @@ -53526,12 +53526,12 @@ "name": "ChatWaifu_v2.0_22B", "developer": "spow12", "scores": { - "IFEval": 0.6517, - "BBH": 0.5908, - "MATH Level 5": 0.2032, - "GPQA": 0.3238, + "IFEval": 0.6511, + "BBH": 0.5926, + "MATH Level 5": 0.1858, + "GPQA": 0.3247, "MUSR": 0.3842, - "MMLU-PRO": 0.3812 + "MMLU-PRO": 0.3836 } }, { @@ -56971,12 +56971,12 @@ "name": "Hebrew-Mistral-7B-200K", "developer": "yam-peleg", "scores": { - "IFEval": 0.177, - "BBH": 0.3411, - "MATH Level 5": 0.031, - "GPQA": 0.2534, - "MUSR": 0.374, - "MMLU-PRO": 0.2529 + "IFEval": 0.1856, + "BBH": 0.4149, + "MATH Level 5": 0.0234, + "GPQA": 0.276, + "MUSR": 0.3765, + "MMLU-PRO": 0.2573 } }, { diff --git a/data/benchmarks/la_leaderboard.json b/data/benchmarks/la_leaderboard.json new file mode 100644 index 0000000000000000000000000000000000000000..3a6324b71798175e00f554ac956af49f5b3647df --- /dev/null +++ b/data/benchmarks/la_leaderboard.json @@ -0,0 +1,44 @@ +{ + "models": [ + { + "model_id": "Qwen/Qwen2.5-7B", + "name": "Qwen2.5-7B", + "developer": "unknown", + "scores": { + "la_leaderboard": 27.61 + } + }, + { + "model_id": "google/gemma-2-9b-it", + "name": "Gemma 2 Instruct 9B", + "developer": "unknown", + "scores": { + "la_leaderboard": 33.62 + } + }, + { + "model_id": "meta-llama/Meta-Llama-3.1-8B", + "name": "Meta Llama 3.1 8B", + "developer": "unknown", + "scores": { + "la_leaderboard": 27.04 + } + }, + { + "model_id": "meta-llama/Meta-Llama-3.1-8B-Instruct", + "name": "Meta Llama 3.1 8B Instruct", + "developer": "unknown", + "scores": { + "la_leaderboard": 30.23 + } + }, + { + "model_id": "utter-project/EuroLLM-9B", + "name": "EuroLLM 9B", + "developer": "unknown", + "scores": { + "la_leaderboard": 25.87 + } + } + ] +} \ No newline at end of file diff --git a/data/benchmarks/livecodebenchpro.json b/data/benchmarks/livecodebenchpro.json index 6792c24758cd5b97dfe6d07eea5a498bc4080cbe..4e2718a62d52e992cfaaada9528b9a430c578ec4 100644 --- a/data/benchmarks/livecodebenchpro.json +++ b/data/benchmarks/livecodebenchpro.json @@ -53,7 +53,7 @@ { "model_id": "anthropic/claude-3-7-sonnet-20250219", "name": "claude-3-7-sonnet-20250219", - "developer": "anthropic", + "developer": "Anthropic", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.0, @@ -143,7 +143,7 @@ { "model_id": "google/gemini-2.5-flash", "name": "gemini-2.5-flash", - "developer": "google", + "developer": "Google", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.028169014084507043, @@ -153,7 +153,7 @@ { "model_id": "google/gemini-2.5-pro", "name": "gemini-2.5-pro", - "developer": "google", + "developer": "Google", "scores": { "Hard Problems": 0.014084507042253521, "Medium Problems": 0.2112676056338028, @@ -193,7 +193,7 @@ { "model_id": "openai/gpt-4o-2024-11-20", "name": "GPT-4o 2024-11-20", - "developer": "openai", + "developer": "OpenAI", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.0, @@ -213,7 +213,7 @@ { "model_id": "openai/gpt-5.2-2025-12-11", "name": "gpt-5.2-2025-12-11", - "developer": "OpenAI", + "developer": "openai", "scores": { "Hard Problems": 0.1594, "Medium Problems": 0.5211, @@ -223,7 +223,7 @@ { "model_id": "openai/gpt-oss-120b", "name": "gpt-oss-120b", - "developer": "openai", + "developer": "OpenAI", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.11267605633802817, @@ -233,7 +233,7 @@ { "model_id": "openai/gpt-oss-20b", "name": "gpt-oss-20b", - "developer": "openai", + "developer": "OpenAI", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.056338028169014086, @@ -243,7 +243,7 @@ { "model_id": "openai/o3-2025-04-16", "name": "o3 2025-04-16", - "developer": "openai", + "developer": "OpenAI", "scores": { "Hard Problems": 0.0, "Medium Problems": 0.22535211267605634, @@ -255,9 +255,9 @@ "name": "o4-mini-2025-04-16", "developer": "openai", "scores": { - "Hard Problems": 0.014084507042253521, - "Medium Problems": 0.30985915492957744, - "Easy Problems": 0.8873239436619719 + "Hard Problems": 0.0143, + "Medium Problems": 0.2923, + "Easy Problems": 0.8571 } }, { diff --git a/data/benchmarks/reward-bench.json b/data/benchmarks/reward-bench.json index e88892a000e33069b0f27e7d091b5b5b1b11b456..2d1697e5fbeb29096457ce7beea667c4c81d3b1d 100644 --- a/data/benchmarks/reward-bench.json +++ b/data/benchmarks/reward-bench.json @@ -131,13 +131,12 @@ "name": "CIR-AMS/BTRM_Qwen2_7b_0613", "developer": "CIR-AMS", "scores": { - "Score": 0.5736, - "Factuality": 0.5347, - "Precise IF": 0.3563, - "Math": 0.6066, - "Safety": 0.7178, - "Focus": 0.5737, - "Ties": 0.6527 + "Score": 0.8172, + "Chat": 0.9749, + "Chat Hard": 0.5724, + "Safety": 0.9014, + "Reasoning": 0.8775, + "Prior Sets (0.5 weight)": 0.7029 } }, { @@ -499,13 +498,11 @@ "name": "LxzGordon/URM-LLaMa-3.1-8B", "developer": "LxzGordon", "scores": { - "Score": 0.7394, - "Factuality": 0.6884, - "Precise IF": 0.45, - "Math": 0.6393, - "Safety": 0.9178, - "Focus": 0.9758, - "Ties": 0.7653 + "Score": 0.9294, + "Chat": 0.9553, + "Chat Hard": 0.8816, + "Safety": 0.9108, + "Reasoning": 0.9698 } }, { @@ -525,11 +522,13 @@ "name": "NCSOFT/Llama-3-OffsetBias-RM-8B", "developer": "NCSOFT", "scores": { - "Score": 0.8942, - "Chat": 0.9721, - "Chat Hard": 0.818, - "Safety": 0.8676, - "Reasoning": 0.9192 + "Score": 0.648, + "Factuality": 0.6084, + "Precise IF": 0.4, + "Math": 0.5191, + "Safety": 0.7222, + "Focus": 0.9596, + "Ties": 0.6786 } }, { @@ -537,13 +536,12 @@ "name": "Nexusflow/Starling-RM-34B", "developer": "Nexusflow", "scores": { - "Score": 0.4553, - "Factuality": 0.4589, - "Precise IF": 0.3187, - "Math": 0.6175, - "Safety": 0.7556, - "Focus": 0.4808, - "Ties": 0.1004 + "Score": 0.8133, + "Chat": 0.9693, + "Chat Hard": 0.5724, + "Safety": 0.877, + "Reasoning": 0.8845, + "Prior Sets (0.5 weight)": 0.7137 } }, { @@ -616,12 +614,13 @@ "name": "OpenAssistant/reward-model-deberta-v3-large-v2", "developer": "OpenAssistant", "scores": { - "Score": 0.6126, - "Chat": 0.8939, - "Chat Hard": 0.4518, - "Safety": 0.7338, - "Reasoning": 0.3855, - "Prior Sets (0.5 weight)": 0.5836 + "Score": 0.32, + "Factuality": 0.3853, + "Precise IF": 0.2687, + "Math": 0.5027, + "Safety": 0.3667, + "Focus": 0.2768, + "Ties": 0.12 } }, { @@ -629,12 +628,13 @@ "name": "PKU-Alignment/beaver-7b-v1.0-cost", "developer": "PKU-Alignment", "scores": { - "Score": 0.5798, - "Chat": 0.6173, - "Chat Hard": 0.4232, - "Safety": 0.7351, - "Reasoning": 0.5482, - "Prior Sets (0.5 weight)": 0.57 + "Score": 0.3332, + "Factuality": 0.3263, + "Precise IF": 0.2313, + "Math": 0.3989, + "Safety": 0.7589, + "Focus": 0.2939, + "Ties": -0.01 } }, { @@ -642,13 +642,12 @@ "name": "PKU-Alignment/beaver-7b-v1.0-reward", "developer": "PKU-Alignment", "scores": { - "Score": 0.1606, - "Factuality": 0.2105, - "Precise IF": 0.2938, - "Math": 0.2623, - "Safety": 0.1422, - "Focus": 0.0646, - "Ties": -0.01 + "Score": 0.4727, + "Chat": 0.8184, + "Chat Hard": 0.2873, + "Safety": 0.3757, + "Reasoning": 0.346, + "Prior Sets (0.5 weight)": 0.5993 } }, { @@ -656,12 +655,13 @@ "name": "PKU-Alignment/beaver-7b-v2.0-cost", "developer": "PKU-Alignment", "scores": { - "Score": 0.5957, - "Chat": 0.5726, - "Chat Hard": 0.4561, - "Safety": 0.7608, - "Reasoning": 0.6211, - "Prior Sets (0.5 weight)": 0.5397 + "Score": 0.3326, + "Factuality": 0.3789, + "Precise IF": 0.275, + "Math": 0.3333, + "Safety": 0.7356, + "Focus": 0.2828, + "Ties": -0.01 } }, { @@ -913,13 +913,11 @@ "name": "Ray2333/GRM-gemma2-2B-rewardmodel-ft", "developer": "Ray2333", "scores": { - "Score": 0.5966, - "Factuality": 0.5305, - "Precise IF": 0.3125, - "Math": 0.5902, - "Safety": 0.9222, - "Focus": 0.7455, - "Ties": 0.4788 + "Score": 0.8839, + "Chat": 0.9302, + "Chat Hard": 0.7719, + "Safety": 0.9216, + "Reasoning": 0.912 } }, { @@ -1077,13 +1075,11 @@ "name": "ShikaiChen/LDL-Reward-Gemma-2-27B-v0.1", "developer": "ShikaiChen", "scores": { - "Score": 0.7249, - "Factuality": 0.7558, - "Precise IF": 0.35, - "Math": 0.6448, - "Safety": 0.9222, - "Focus": 0.9131, - "Ties": 0.7633 + "Score": 0.9499, + "Chat": 0.9637, + "Chat Hard": 0.9079, + "Safety": 0.9378, + "Reasoning": 0.9903 } }, { @@ -1115,13 +1111,11 @@ "name": "Skywork/Skywork-Reward-Gemma-2-27B", "developer": "Skywork", "scores": { - "Score": 0.7576, - "Factuality": 0.7368, - "Precise IF": 0.4031, - "Math": 0.7049, - "Safety": 0.9422, - "Focus": 0.9323, - "Ties": 0.8261 + "Score": 0.938, + "Chat": 0.9581, + "Chat Hard": 0.9145, + "Safety": 0.9189, + "Reasoning": 0.9606 } }, { @@ -1340,10 +1334,10 @@ "name": "ai2/tulu-2-7b-rm-v0-nectar-binarized-3.8m-check...", "developer": "ai2", "scores": { - "Score": 0.7058, - "Chat": 0.9525, - "Chat Hard": 0.3947, - "Safety": 0.7703 + "Score": 0.6905, + "Chat": 0.9441, + "Chat Hard": 0.3596, + "Safety": 0.7676 } }, { @@ -1384,13 +1378,12 @@ "name": "allenai/Llama-3.1-70B-Instruct-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.7606, - "Factuality": 0.8126, - "Precise IF": 0.4188, - "Math": 0.6995, - "Safety": 0.8844, - "Focus": 0.8646, - "Ties": 0.8835 + "Score": 0.9021, + "Chat": 0.9665, + "Chat Hard": 0.8355, + "Safety": 0.9095, + "Reasoning": 0.8969, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1398,12 +1391,13 @@ "name": "allenai/Llama-3.1-8B-Base-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8463, - "Chat": 0.933, - "Chat Hard": 0.7785, - "Safety": 0.8851, - "Reasoning": 0.7886, - "Prior Sets (0.5 weight)": 0.0 + "Score": 0.649, + "Factuality": 0.72, + "Precise IF": 0.3625, + "Math": 0.612, + "Safety": 0.8267, + "Focus": 0.8323, + "Ties": 0.5406 } }, { @@ -1411,13 +1405,12 @@ "name": "allenai/Llama-3.1-8B-Instruct-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.7285, - "Factuality": 0.7432, - "Precise IF": 0.4437, - "Math": 0.6175, - "Safety": 0.8956, - "Focus": 0.9071, - "Ties": 0.7638 + "Score": 0.8885, + "Chat": 0.9581, + "Chat Hard": 0.8158, + "Safety": 0.8932, + "Reasoning": 0.887, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1425,13 +1418,12 @@ "name": "allenai/Llama-3.1-Tulu-3-70B-SFT-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.722, - "Factuality": 0.8084, - "Precise IF": 0.3688, - "Math": 0.6776, - "Safety": 0.8689, - "Focus": 0.7778, - "Ties": 0.8308 + "Score": 0.8892, + "Chat": 0.9693, + "Chat Hard": 0.8268, + "Safety": 0.9027, + "Reasoning": 0.8583, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -3673,13 +3665,12 @@ "name": "hendrydong/Mistral-RM-for-RAFT-GSHF-v0", "developer": "hendrydong", "scores": { - "Score": 0.5851, - "Factuality": 0.5779, - "Precise IF": 0.3625, - "Math": 0.6011, - "Safety": 0.6956, - "Focus": 0.6747, - "Ties": 0.5988 + "Score": 0.7847, + "Chat": 0.9832, + "Chat Hard": 0.5789, + "Safety": 0.85, + "Reasoning": 0.7434, + "Prior Sets (0.5 weight)": 0.7508 } }, { @@ -3687,13 +3678,11 @@ "name": "infly/INF-ORM-Llama3.1-70B", "developer": "infly", "scores": { - "Score": 0.7648, - "Factuality": 0.7411, - "Precise IF": 0.4188, - "Math": 0.6995, - "Safety": 0.9644, - "Focus": 0.903, - "Ties": 0.8622 + "Score": 0.9511, + "Chat": 0.9665, + "Chat Hard": 0.9101, + "Safety": 0.9365, + "Reasoning": 0.9912 } }, { @@ -3701,11 +3690,13 @@ "name": "internlm/internlm2-1_8b-reward", "developer": "internlm", "scores": { - "Score": 0.8217, - "Chat": 0.9358, - "Chat Hard": 0.6623, - "Safety": 0.8162, - "Reasoning": 0.8724 + "Score": 0.3902, + "Factuality": 0.2758, + "Precise IF": 0.3625, + "Math": 0.4426, + "Safety": 0.4711, + "Focus": 0.596, + "Ties": 0.1934 } }, { @@ -3713,11 +3704,13 @@ "name": "internlm/internlm2-20b-reward", "developer": "internlm", "scores": { - "Score": 0.9016, - "Chat": 0.9888, - "Chat Hard": 0.7654, - "Safety": 0.8946, - "Reasoning": 0.9576 + "Score": 0.5628, + "Factuality": 0.5558, + "Precise IF": 0.3625, + "Math": 0.5738, + "Safety": 0.6111, + "Focus": 0.7253, + "Ties": 0.5483 } }, { @@ -3901,13 +3894,11 @@ "name": "nicolinho/QRM-Gemma-2-27B", "developer": "nicolinho", "scores": { - "Score": 0.7667, - "Factuality": 0.7853, - "Precise IF": 0.3719, - "Math": 0.6995, - "Safety": 0.9578, - "Focus": 0.9535, - "Ties": 0.8321 + "Score": 0.9444, + "Chat": 0.9665, + "Chat Hard": 0.9013, + "Safety": 0.927, + "Reasoning": 0.9826 } }, { @@ -3939,13 +3930,11 @@ "name": "nicolinho/QRM-Llama3.1-8B-v2", "developer": "nicolinho", "scores": { - "Score": 0.7074, - "Factuality": 0.6653, - "Precise IF": 0.4062, - "Math": 0.612, - "Safety": 0.9467, - "Focus": 0.8909, - "Ties": 0.7234 + "Score": 0.9314, + "Chat": 0.9637, + "Chat Hard": 0.8684, + "Safety": 0.9257, + "Reasoning": 0.9677 } }, { @@ -4097,13 +4086,11 @@ "name": "GPT-4o mini 2024-07-18", "developer": "openai", "scores": { - "Score": 0.5796, - "Factuality": 0.4105, - "Precise IF": 0.3438, - "Math": 0.5191, - "Safety": 0.7667, - "Focus": 0.7414, - "Ties": 0.6962 + "Score": 0.8007, + "Chat": 0.9497, + "Chat Hard": 0.6075, + "Safety": 0.8081, + "Reasoning": 0.8374 } }, { @@ -4124,12 +4111,13 @@ "name": "openbmb/Eurus-RM-7b", "developer": "openbmb", "scores": { - "Score": 0.8159, - "Chat": 0.9804, - "Chat Hard": 0.6557, - "Safety": 0.8135, - "Reasoning": 0.8633, - "Prior Sets (0.5 weight)": 0.7172 + "Score": 0.5806, + "Factuality": 0.6, + "Precise IF": 0.3438, + "Math": 0.5683, + "Safety": 0.6267, + "Focus": 0.7475, + "Ties": 0.5972 } }, { @@ -4150,13 +4138,12 @@ "name": "openbmb/UltraRM-13b", "developer": "openbmb", "scores": { - "Score": 0.4683, - "Factuality": 0.5063, - "Precise IF": 0.3312, - "Math": 0.5519, - "Safety": 0.5089, - "Focus": 0.6081, - "Ties": 0.3036 + "Score": 0.6903, + "Chat": 0.9637, + "Chat Hard": 0.5548, + "Safety": 0.5986, + "Reasoning": 0.6244, + "Prior Sets (0.5 weight)": 0.7294 } }, { @@ -4394,13 +4381,12 @@ "name": "weqweasdas/RM-Mistral-7B", "developer": "weqweasdas", "scores": { - "Score": 0.596, - "Factuality": 0.5937, - "Precise IF": 0.3438, - "Math": 0.5956, - "Safety": 0.6911, - "Focus": 0.7293, - "Ties": 0.6226 + "Score": 0.7982, + "Chat": 0.9665, + "Chat Hard": 0.6053, + "Safety": 0.8703, + "Reasoning": 0.7736, + "Prior Sets (0.5 weight)": 0.753 } }, { diff --git a/data/benchmarks/swe-bench.json b/data/benchmarks/swe-bench.json index eb8b39879e565a2a2d717362ca5d2e5f29196eee..5ffba3c27cbc351e30494b9623bfb1c1b9296cb6 100644 --- a/data/benchmarks/swe-bench.json +++ b/data/benchmarks/swe-bench.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "swe-bench": 0.6061 + "swe-bench": 0.7423 } }, { diff --git a/data/benchmarks/tau-bench-2_retail.json b/data/benchmarks/tau-bench-2_retail.json index 2da1258ba59e3f413441ef0b6df6faabeb6eb792..20cbcadcbc839a7385e1150d11b5c54f3213804c 100644 --- a/data/benchmarks/tau-bench-2_retail.json +++ b/data/benchmarks/tau-bench-2_retail.json @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/retail": 0.68 + "tau-bench-2/retail": 0.5354 } } ] diff --git a/data/benchmarks/tau-bench-2_telecom.json b/data/benchmarks/tau-bench-2_telecom.json index 5e2e97c5a63c814404bfd0e936bb7f41ce63593e..3f23a59ee9015ad312e50ba482415cf77ef024b0 100644 --- a/data/benchmarks/tau-bench-2_telecom.json +++ b/data/benchmarks/tau-bench-2_telecom.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/telecom": 0.73 + "tau-bench-2/telecom": 0.88 } }, { diff --git a/data/benchmarks/terminal-bench-2.0.json b/data/benchmarks/terminal-bench-2.0.json index 1b7467dde0ee56ac12d5ecd84baf3cf468c0942e..7972b2156ba111197822f88d0841c7ebf9f8bf1d 100644 --- a/data/benchmarks/terminal-bench-2.0.json +++ b/data/benchmarks/terminal-bench-2.0.json @@ -13,7 +13,7 @@ "name": "Claude Haiku 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 29.8 + "terminal-bench-2.0": 28.3 } }, { @@ -29,7 +29,7 @@ "name": "Claude Opus 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 59.1 + "terminal-bench-2.0": 51.7 } }, { @@ -37,7 +37,7 @@ "name": "Claude Opus 4.6", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 58.0 + "terminal-bench-2.0": 69.9 } }, { @@ -45,7 +45,7 @@ "name": "Claude Sonnet 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 46.5 + "terminal-bench-2.0": 42.8 } }, { @@ -59,7 +59,7 @@ { "model_id": "google/gemini-2.5-flash", "name": "gemini-2.5-flash", - "developer": "google", + "developer": "Google", "scores": { "terminal-bench-2.0": 17.1 } @@ -67,9 +67,9 @@ { "model_id": "google/gemini-2.5-pro", "name": "gemini-2.5-pro", - "developer": "google", + "developer": "Google", "scores": { - "terminal-bench-2.0": 26.1 + "terminal-bench-2.0": 19.6 } }, { @@ -77,7 +77,7 @@ "name": "Gemini 3 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 64.3 + "terminal-bench-2.0": 51.7 } }, { @@ -93,7 +93,7 @@ "name": "Gemini 3.1 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 74.8 + "terminal-bench-2.0": 78.4 } }, { @@ -109,7 +109,7 @@ "name": "MiniMax M2.1", "developer": "MiniMax", "scores": { - "terminal-bench-2.0": 36.6 + "terminal-bench-2.0": 29.2 } }, { @@ -149,7 +149,7 @@ "name": "Multiple", "developer": "Multiple", "scores": { - "terminal-bench-2.0": 59.1 + "terminal-bench-2.0": 58.4 } }, { @@ -157,7 +157,7 @@ "name": "GPT-5", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 33.9 + "terminal-bench-2.0": 49.6 } }, { @@ -173,7 +173,7 @@ "name": "GPT-5-Mini", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 29.2 + "terminal-bench-2.0": 24.0 } }, { @@ -181,7 +181,7 @@ "name": "GPT-5-Nano", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 9.9 + "terminal-bench-2.0": 11.5 } }, { @@ -221,7 +221,7 @@ "name": "GPT-5.2", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 54.0 + "terminal-bench-2.0": 62.9 } }, { @@ -243,15 +243,15 @@ { "model_id": "openai/gpt-oss-120b", "name": "gpt-oss-120b", - "developer": "openai", + "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 14.2 + "terminal-bench-2.0": 18.7 } }, { "model_id": "openai/gpt-oss-20b", "name": "gpt-oss-20b", - "developer": "openai", + "developer": "OpenAI", "scores": { "terminal-bench-2.0": 3.1 } @@ -261,7 +261,7 @@ "name": "Grok 4", "developer": "xAI", "scores": { - "terminal-bench-2.0": 25.4 + "terminal-bench-2.0": 27.2 } }, { @@ -269,7 +269,7 @@ "name": "Grok Code Fast 1", "developer": "xAI", "scores": { - "terminal-bench-2.0": 25.8 + "terminal-bench-2.0": 14.2 } }, { diff --git a/data/developers.json b/data/developers.json index d2b684e9fa520000f43c13473bfd59f34c8ac7b5..4f49921e3a331c23a76fdf8cced7196b015f8063 100644 --- a/data/developers.json +++ b/data/developers.json @@ -32,1399 +32,103 @@ "model_count": 1 }, { - "developer": "AALF", - "model_count": 4 - }, - { - "developer": "AELLM", - "model_count": 2 - }, - { - "developer": "AGI-0", - "model_count": 3 - }, - { - "developer": "AI-MO", - "model_count": 2 - }, - { - "developer": "AI-Sweden-Models", - "model_count": 2 - }, - { - "developer": "AI4free", - "model_count": 2 - }, - { - "developer": "AIDC-AI", - "model_count": 1 - }, - { - "developer": "Aashraf995", - "model_count": 4 - }, - { - "developer": "AbacusResearch", - "model_count": 1 - }, - { - "developer": "Ahdoot", - "model_count": 2 - }, - { - "developer": "Ahjeong", - "model_count": 2 - }, - { - "developer": "AicoresSecurity", - "model_count": 4 - }, - { - "developer": "Alepach", - "model_count": 3 - }, - { - "developer": "AlephAlpha", - "model_count": 3 - }, - { - "developer": "Alibaba", - "model_count": 5 - }, - { - "developer": "Alibaba-NLP", - "model_count": 1 - }, - { - "developer": "Alsebay", - "model_count": 1 - }, - { - "developer": "Amaorynho", - "model_count": 4 - }, - { - "developer": "Amu", - "model_count": 2 - }, - { - "developer": "Anthropic", - "model_count": 11 - }, - { - "developer": "ArliAI", - "model_count": 2 - }, - { - "developer": "Arthur-LAGACHERIE", - "model_count": 1 - }, - { - "developer": "Artples", - "model_count": 2 - }, - { - "developer": "Aryanne", - "model_count": 3 - }, - { - "developer": "AtAndDev", - "model_count": 1 - }, - { - "developer": "Ateron", - "model_count": 3 - }, - { - "developer": "AtlaAI", - "model_count": 2 - }, - { - "developer": "AuraIndustries", - "model_count": 4 - }, - { - "developer": "Aurel9", - "model_count": 1 - }, - { - "developer": "Ayush-Singh", - "model_count": 1 - }, - { - "developer": "Azure99", - "model_count": 6 - }, - { - "developer": "BAAI", - "model_count": 14 - }, - { - "developer": "BEE-spoke-data", - "model_count": 9 - }, - { - "developer": "BSC-LT", - "model_count": 2 - }, - { - "developer": "Ba2han", - "model_count": 1 - }, - { - "developer": "Baptiste-HUVELLE-10", - "model_count": 1 - }, - { - "developer": "BenevolenceMessiah", - "model_count": 2 - }, - { - "developer": "BlackBeenie", - "model_count": 9 - }, - { - "developer": "Bllossom", - "model_count": 1 - }, - { - "developer": "BoltMonkey", - "model_count": 3 - }, - { - "developer": "BrainWave-ML", - "model_count": 1 - }, - { - "developer": "BramVanroy", - "model_count": 4 - }, - { - "developer": "ByteDance", - "model_count": 1 - }, - { - "developer": "CIR-AMS", - "model_count": 1 - }, - { - "developer": "CYFRAGOVPL", - "model_count": 6 - }, - { - "developer": "CarrotAI", - "model_count": 2 - }, - { - "developer": "Casual-Autopsy", - "model_count": 1 - }, - { - "developer": "CausalLM", - "model_count": 3 - }, - { - "developer": "Changgil", - "model_count": 2 - }, - { - "developer": "ClaudioItaly", - "model_count": 4 - }, - { - "developer": "CohereForAI", - "model_count": 8 - }, - { - "developer": "Columbia-NLP", - "model_count": 6 - }, - { - "developer": "CombinHorizon", - "model_count": 6 - }, - { - "developer": "ContactDoctor", - "model_count": 2 - }, - { - "developer": "ContextualAI", - "model_count": 16 - }, - { - "developer": "CoolSpring", - "model_count": 3 - }, - { - "developer": "Corianas", - "model_count": 3 - }, - { - "developer": "CortexLM", - "model_count": 1 - }, - { - "developer": "Cran-May", - "model_count": 7 - }, - { - "developer": "CreitinGameplays", - "model_count": 1 - }, - { - "developer": "CultriX", - "model_count": 32 - }, - { - "developer": "DRXD1000", - "model_count": 2 - }, - { - "developer": "DUAL-GPO", - "model_count": 1 - }, - { - "developer": "DZgas", - "model_count": 1 - }, - { - "developer": "Daemontatox", - "model_count": 32 - }, - { - "developer": "Dampfinchen", - "model_count": 1 - }, - { - "developer": "Danielbrdz", - "model_count": 7 - }, - { - "developer": "Dans-DiscountModels", - "model_count": 9 - }, - { - "developer": "Darkknight535", - "model_count": 1 - }, - { - "developer": "Databricks-Mosaic-Research", - "model_count": 1 - }, - { - "developer": "DavidAU", - "model_count": 25 - }, - { - "developer": "Davidsv", - "model_count": 1 - }, - { - "developer": "DavieLion", - "model_count": 5 - }, - { - "developer": "DebateLabKIT", - "model_count": 1 - }, - { - "developer": "Deci", - "model_count": 2 - }, - { - "developer": "DeepAutoAI", - "model_count": 12 - }, - { - "developer": "DeepMount00", - "model_count": 14 - }, - { - "developer": "DeepSeek", - "model_count": 5 - }, - { - "developer": "Delta-Vector", - "model_count": 7 - }, - { - "developer": "DevQuasar", - "model_count": 1 - }, - { - "developer": "Dongwei", - "model_count": 1 - }, - { - "developer": "DoppelReflEx", - "model_count": 29 - }, - { - "developer": "DreadPoor", - "model_count": 119 - }, - { - "developer": "ECE-ILAB-PRYMMAL", - "model_count": 1 - }, - { - "developer": "EVA-UNIT-01", - "model_count": 2 - }, - { - "developer": "Edgerunners", - "model_count": 1 - }, - { - "developer": "EleutherAI", - "model_count": 12 - }, - { - "developer": "Enno-Ai", - "model_count": 4 - }, - { - "developer": "EnnoAi", - "model_count": 2 - }, - { - "developer": "Epiculous", - "model_count": 4 - }, - { - "developer": "EpistemeAI", - "model_count": 47 - }, - { - "developer": "EpistemeAI2", - "model_count": 15 - }, - { - "developer": "Eric111", - "model_count": 2 - }, - { - "developer": "Etherll", - "model_count": 8 - }, - { - "developer": "Eurdem", - "model_count": 1 - }, - { - "developer": "FINGU-AI", - "model_count": 7 - }, - { - "developer": "FallenMerick", - "model_count": 1 - }, - { - "developer": "Felladrin", - "model_count": 2 - }, - { - "developer": "FlofloB", - "model_count": 27 - }, - { - "developer": "FuJhen", - "model_count": 4 - }, - { - "developer": "FuseAI", - "model_count": 4 - }, - { - "developer": "GalrionSoftworks", - "model_count": 2 - }, - { - "developer": "GenVRadmin", - "model_count": 4 - }, - { - "developer": "GoToCompany", - "model_count": 2 - }, - { - "developer": "Goekdeniz-Guelmez", - "model_count": 10 - }, - { - "developer": "Google", - "model_count": 4 - }, - { - "developer": "GreenNode", - "model_count": 1 - }, - { - "developer": "GritLM", - "model_count": 2 - }, - { - "developer": "Groq", - "model_count": 1 - }, - { - "developer": "Gryphe", - "model_count": 5 - }, - { - "developer": "GuilhermeNaturaUmana", - "model_count": 1 - }, - { - "developer": "Gunulhona", - "model_count": 2 - }, - { - "developer": "HFXM", - "model_count": 1 - }, - { - "developer": "HPAI-BSC", - "model_count": 3 - }, - { - "developer": "HarbingerX", - "model_count": 2 - }, - { - "developer": "Hastagaras", - "model_count": 3 - }, - { - "developer": "HelpingAI", - "model_count": 4 - }, - { - "developer": "HeraiHench", - "model_count": 4 - }, - { - "developer": "HiroseKoichi", - "model_count": 1 - }, - { - "developer": "HoangHa", - "model_count": 1 - }, - { - "developer": "HuggingFaceH4", - "model_count": 5 - }, - { - "developer": "HuggingFaceTB", - "model_count": 12 - }, - { - "developer": "HumanLLMs", - "model_count": 3 - }, - { - "developer": "IDEA-CCNL", - "model_count": 2 - }, - { - "developer": "INSAIT-Institute", - "model_count": 1 - }, - { - "developer": "IlyaGusev", - "model_count": 2 - }, - { - "developer": "Infinirc", - "model_count": 1 - }, - { - "developer": "Intel", - "model_count": 4 - }, - { - "developer": "IntervitensInc", - "model_count": 1 - }, - { - "developer": "Invalid-Null", - "model_count": 2 - }, - { - "developer": "Isaak-Carter", - "model_count": 3 - }, - { - "developer": "J-LAB", - "model_count": 1 - }, - { - "developer": "JackFram", - "model_count": 2 - }, - { - "developer": "Jacoby746", - "model_count": 7 - }, - { - "developer": "JayHyeon", - "model_count": 174 - }, - { - "developer": "Jimmy19991222", - "model_count": 8 - }, - { - "developer": "Joseph717171", - "model_count": 2 - }, - { - "developer": "Josephgflowers", - "model_count": 7 - }, - { - "developer": "JungZoona", - "model_count": 2 - }, - { - "developer": "Junhoee", - "model_count": 1 - }, - { - "developer": "KSU-HW-SEC", - "model_count": 4 - }, - { - "developer": "Khetterman", - "model_count": 2 - }, - { - "developer": "Kimargin", - "model_count": 1 - }, - { - "developer": "Kimi", - "model_count": 1 - }, - { - "developer": "KingNish", - "model_count": 7 - }, - { - "developer": "Kquant03", - "model_count": 2 - }, - { - "developer": "Krystalan", - "model_count": 2 - }, - { - "developer": "Kuaishou", - "model_count": 1 - }, - { - "developer": "Kukedlc", - "model_count": 7 - }, - { - "developer": "Kumar955", - "model_count": 1 - }, - { - "developer": "L-RAGE", - "model_count": 1 - }, - { - "developer": "LEESM", - "model_count": 4 - }, - { - "developer": "LGAI-EXAONE", - "model_count": 4 - }, - { - "developer": "LLM360", - "model_count": 2 - }, - { - "developer": "LLM4Binary", - "model_count": 1 - }, - { - "developer": "Lambent", - "model_count": 1 - }, - { - "developer": "Langboat", - "model_count": 1 - }, - { - "developer": "Lawnakk", - "model_count": 10 - }, - { - "developer": "LenguajeNaturalAI", - "model_count": 2 - }, - { - "developer": "LeroyDyer", - "model_count": 58 - }, - { - "developer": "LightningRodLabs", - "model_count": 3 - }, - { - "developer": "Lil-R", - "model_count": 8 - }, - { - "developer": "LilRg", - "model_count": 10 - }, - { - "developer": "LimYeri", - "model_count": 5 - }, - { - "developer": "Locutusque", - "model_count": 6 - }, - { - "developer": "Luni", - "model_count": 2 - }, - { - "developer": "Lunzima", - "model_count": 18 - }, - { - "developer": "LxzGordon", - "model_count": 2 - }, - { - "developer": "Lyte", - "model_count": 3 - }, - { - "developer": "M4-ai", - "model_count": 1 - }, - { - "developer": "MEscriva", - "model_count": 1 - }, - { - "developer": "MLP-KTLim", - "model_count": 1 - }, - { - "developer": "MTSAIR", - "model_count": 2 - }, - { - "developer": "Magpie-Align", - "model_count": 8 - }, - { - "developer": "MagusCorp", - "model_count": 1 - }, - { - "developer": "ManoloPueblo", - "model_count": 3 - }, - { - "developer": "MarinaraSpaghetti", - "model_count": 2 - }, - { - "developer": "Marsouuu", - "model_count": 8 - }, - { - "developer": "MaziyarPanahi", - "model_count": 44 - }, - { - "developer": "Meta", - "model_count": 1 - }, - { - "developer": "Minami-su", - "model_count": 5 - }, - { - "developer": "MiniMax", - "model_count": 2 - }, - { - "developer": "Minimax", - "model_count": 1 - }, - { - "developer": "ModelCloud", - "model_count": 1 - }, - { - "developer": "ModelSpace", - "model_count": 1 - }, - { - "developer": "MoonRide", - "model_count": 1 - }, - { - "developer": "Moonshot AI", - "model_count": 2 - }, - { - "developer": "Mostafa8Mehrabi", - "model_count": 1 - }, - { - "developer": "MrRobotoAI", - "model_count": 2 - }, - { - "developer": "Multiple", - "model_count": 1 - }, - { - "developer": "MultivexAI", - "model_count": 5 - }, - { - "developer": "Mxode", - "model_count": 5 - }, - { - "developer": "NAPS-ai", - "model_count": 7 - }, - { - "developer": "NCSOFT", - "model_count": 3 - }, - { - "developer": "NJS26", - "model_count": 1 - }, - { - "developer": "NLPark", - "model_count": 3 - }, - { - "developer": "NTQAI", - "model_count": 2 - }, - { - "developer": "NYTK", - "model_count": 2 - }, - { - "developer": "Naveenpoliasetty", - "model_count": 1 - }, - { - "developer": "NbAiLab", - "model_count": 2 - }, - { - "developer": "Nekochu", - "model_count": 4 - }, - { - "developer": "NeverSleep", - "model_count": 2 - }, - { - "developer": "Nexesenex", - "model_count": 45 - }, - { - "developer": "Nexusflow", - "model_count": 2 - }, - { - "developer": "NikolaSigmoid", - "model_count": 7 - }, - { - "developer": "Nitral-AI", - "model_count": 8 - }, - { - "developer": "Nohobby", - "model_count": 2 - }, - { - "developer": "Norquinal", - "model_count": 8 - }, - { - "developer": "NotASI", - "model_count": 4 - }, - { - "developer": "NousResearch", - "model_count": 19 - }, - { - "developer": "Novaciano", - "model_count": 11 - }, - { - "developer": "NucleusAI", - "model_count": 1 - }, - { - "developer": "NyxKrage", - "model_count": 1 - }, - { - "developer": "OEvortex", - "model_count": 5 - }, - { - "developer": "OliveiraJLT", - "model_count": 1 - }, - { - "developer": "Omkar1102", - "model_count": 1 - }, - { - "developer": "OmnicromsBrain", - "model_count": 1 - }, - { - "developer": "OnlyCheeini", - "model_count": 1 - }, - { - "developer": "Open-Orca", - "model_count": 1 - }, - { - "developer": "OpenAI", - "model_count": 13 - }, - { - "developer": "OpenAssistant", - "model_count": 4 - }, - { - "developer": "OpenBuddy", - "model_count": 22 - }, - { - "developer": "OpenGenerativeAI", - "model_count": 2 - }, - { - "developer": "OpenLLM-France", - "model_count": 4 - }, - { - "developer": "OpenLeecher", - "model_count": 1 - }, - { - "developer": "OpenScholar", - "model_count": 1 - }, - { - "developer": "Orenguteng", - "model_count": 2 - }, - { - "developer": "Orion-zhen", - "model_count": 2 - }, - { - "developer": "P0x0", - "model_count": 1 - }, - { - "developer": "PJMixers", - "model_count": 1 - }, - { - "developer": "PJMixers-Dev", - "model_count": 9 - }, - { - "developer": "PKU-Alignment", - "model_count": 4 - }, - { - "developer": "Parissa3", - "model_count": 1 - }, - { - "developer": "Pinkstack", - "model_count": 4 - }, - { - "developer": "PoLL", - "model_count": 1 - }, - { - "developer": "PocketDoc", - "model_count": 5 - }, - { - "developer": "PowerInfer", - "model_count": 1 - }, - { - "developer": "PranavHarshan", - "model_count": 2 - }, - { - "developer": "Pretergeek", - "model_count": 9 - }, - { - "developer": "PrimeIntellect", - "model_count": 2 - }, - { - "developer": "PuxAI", - "model_count": 1 - }, - { - "developer": "PygmalionAI", - "model_count": 1 - }, - { - "developer": "Q-bert", - "model_count": 1 - }, - { - "developer": "Quazim0t0", - "model_count": 70 - }, - { - "developer": "Qwen", - "model_count": 60 - }, - { - "developer": "R-I-S-E", - "model_count": 2 - }, - { - "developer": "RDson", - "model_count": 1 - }, - { - "developer": "RESMPDEV", - "model_count": 2 - }, - { - "developer": "RLHFlow", - "model_count": 4 - }, - { - "developer": "RWKV", - "model_count": 1 - }, - { - "developer": "Rakuten", - "model_count": 3 - }, - { - "developer": "Ray2333", - "model_count": 10 - }, - { - "developer": "Replete-AI", - "model_count": 9 - }, - { - "developer": "RezVortex", - "model_count": 2 - }, - { - "developer": "Ro-xe", - "model_count": 4 - }, - { - "developer": "Rombo-Org", - "model_count": 1 - }, - { - "developer": "RubielLabarta", - "model_count": 1 - }, - { - "developer": "SF-Foundation", - "model_count": 2 - }, - { - "developer": "SaisExperiments", - "model_count": 6 - }, - { - "developer": "Sakalti", - "model_count": 66 - }, - { - "developer": "Salesforce", - "model_count": 4 - }, - { - "developer": "SanjiWatsuki", - "model_count": 2 - }, - { - "developer": "Sao10K", - "model_count": 8 - }, - { - "developer": "Saxo", - "model_count": 11 - }, - { - "developer": "Schrieffer", - "model_count": 1 - }, - { - "developer": "SeaLLMs", - "model_count": 3 - }, - { - "developer": "SenseLLM", - "model_count": 2 - }, - { - "developer": "SentientAGI", - "model_count": 2 - }, - { - "developer": "SeppeV", - "model_count": 1 - }, - { - "developer": "Sharathhebbar24", - "model_count": 2 - }, - { - "developer": "ShikaiChen", - "model_count": 1 - }, - { - "developer": "Shreyash2010", - "model_count": 1 - }, - { - "developer": "Sicarius-Prototyping", - "model_count": 3 - }, - { - "developer": "SicariusSicariiStuff", - "model_count": 19 - }, - { - "developer": "SkyOrbis", - "model_count": 12 - }, - { - "developer": "Skywork", - "model_count": 15 - }, - { - "developer": "Solshine", - "model_count": 2 - }, - { - "developer": "Sorawiz", - "model_count": 2 - }, - { - "developer": "Sourjayon", - "model_count": 2 - }, - { - "developer": "SpaceYL", - "model_count": 1 - }, - { - "developer": "Spestly", - "model_count": 3 - }, - { - "developer": "Stark2008", - "model_count": 3 - }, - { - "developer": "Steelskull", - "model_count": 2 - }, - { - "developer": "StelleX", - "model_count": 2 - }, - { - "developer": "SultanR", - "model_count": 4 - }, - { - "developer": "Supichi", - "model_count": 11 - }, - { - "developer": "Svak", - "model_count": 2 - }, - { - "developer": "Syed-Hasan-8503", - "model_count": 1 - }, - { - "developer": "T145", - "model_count": 51 - }, - { - "developer": "THUDM", - "model_count": 5 - }, - { - "developer": "TIGER-Lab", - "model_count": 6 - }, - { - "developer": "TTTXXX01", - "model_count": 1 - }, - { - "developer": "Tarek07", - "model_count": 2 - }, - { - "developer": "TeeZee", - "model_count": 1 - }, - { - "developer": "Telugu-LLM-Labs", - "model_count": 2 - }, - { - "developer": "TencentARC", - "model_count": 4 - }, - { - "developer": "TheDrummer", - "model_count": 9 - }, - { - "developer": "TheDrunkenSnail", - "model_count": 3 - }, - { - "developer": "TheHierophant", - "model_count": 1 - }, - { - "developer": "TheTsar1209", - "model_count": 7 - }, - { - "developer": "Tijmen2", - "model_count": 1 - }, - { - "developer": "TinyLlama", - "model_count": 6 - }, - { - "developer": "ToastyPigeon", - "model_count": 1 - }, - { - "developer": "Trappu", - "model_count": 2 - }, - { - "developer": "Tremontaine", + "developer": "aaditya", "model_count": 1 }, { - "developer": "Triangle104", - "model_count": 61 + "developer": "AALF", + "model_count": 4 }, { - "developer": "Tsunami-th", + "developer": "Aashraf995", "model_count": 4 }, { - "developer": "UCLA-AGI", + "developer": "abacusai", "model_count": 10 }, { - "developer": "UKzExecution", - "model_count": 1 - }, - { - "developer": "Unbabel", + "developer": "AbacusResearch", "model_count": 1 }, { - "developer": "Undi95", - "model_count": 2 + "developer": "abhishek", + "model_count": 5 }, { - "developer": "V3N0M", + "developer": "abideen", "model_count": 1 }, { - "developer": "VAGOsolutions", - "model_count": 17 - }, - { - "developer": "VIRNECT", - "model_count": 2 - }, - { - "developer": "ValiantLabs", - "model_count": 11 - }, - { - "developer": "Vikhrmodels", - "model_count": 2 - }, - { - "developer": "Weyaxi", - "model_count": 8 - }, - { - "developer": "WizardLMTeam", - "model_count": 3 - }, - { - "developer": "Wladastic", + "developer": "adamo1139", "model_count": 1 }, { - "developer": "Xclbr7", - "model_count": 4 - }, - { - "developer": "Xiaojian9992024", - "model_count": 12 - }, - { - "developer": "Xkev", + "developer": "adriszmar", "model_count": 1 }, { - "developer": "YOYO-AI", - "model_count": 21 + "developer": "AELLM", + "model_count": 2 }, { - "developer": "Yash21", + "developer": "aevalone", "model_count": 1 }, { - "developer": "Youlln", - "model_count": 19 - }, - { - "developer": "YoungPanda", - "model_count": 1 + "developer": "agentlans", + "model_count": 9 }, { - "developer": "Yuma42", + "developer": "AGI-0", "model_count": 3 }, { - "developer": "Z-AI", + "developer": "Ahdoot", "model_count": 2 }, { - "developer": "Z.AI", - "model_count": 1 - }, - { - "developer": "Z.ai", - "model_count": 1 - }, - { - "developer": "Z1-Coder", - "model_count": 1 - }, - { - "developer": "ZHLiu627", + "developer": "Ahjeong", "model_count": 2 }, { - "developer": "ZeroXClem", - "model_count": 11 - }, - { - "developer": "ZeusLabs", - "model_count": 1 - }, - { - "developer": "ZhangShenao", - "model_count": 1 - }, - { - "developer": "ZiyiYe", - "model_count": 1 - }, - { - "developer": "aaditya", + "developer": "ahmeda335", "model_count": 1 }, { - "developer": "abacusai", - "model_count": 10 - }, - { - "developer": "abhishek", - "model_count": 5 + "developer": "AI-MO", + "model_count": 2 }, { - "developer": "abideen", - "model_count": 1 + "developer": "AI-Sweden-Models", + "model_count": 2 }, { - "developer": "adamo1139", - "model_count": 1 + "developer": "AI2", + "model_count": 7 }, { - "developer": "adriszmar", - "model_count": 1 + "developer": "ai21", + "model_count": 12 }, { - "developer": "aevalone", + "developer": "ai21labs", "model_count": 1 }, { - "developer": "agentlans", - "model_count": 9 - }, - { - "developer": "ahmeda335", + "developer": "ai4bharat", "model_count": 1 }, { - "developer": "ai2", - "model_count": 7 - }, - { - "developer": "ai21", - "model_count": 12 + "developer": "AI4free", + "model_count": 2 }, { - "developer": "ai21labs", - "model_count": 1 + "developer": "AicoresSecurity", + "model_count": 4 }, { - "developer": "ai4bharat", + "developer": "AIDC-AI", "model_count": 1 }, { @@ -1443,12 +147,24 @@ "developer": "alcholjung", "model_count": 1 }, + { + "developer": "Alepach", + "model_count": 3 + }, { "developer": "aleph-alpha", "model_count": 3 }, { - "developer": "alibaba", + "developer": "AlephAlpha", + "model_count": 3 + }, + { + "developer": "Alibaba", + "model_count": 6 + }, + { + "developer": "Alibaba-NLP", "model_count": 1 }, { @@ -1475,10 +191,18 @@ "developer": "alpindale", "model_count": 2 }, + { + "developer": "Alsebay", + "model_count": 1 + }, { "developer": "altomek", "model_count": 1 }, + { + "developer": "Amaorynho", + "model_count": 4 + }, { "developer": "amazon", "model_count": 5 @@ -1487,6 +211,10 @@ "developer": "amd", "model_count": 1 }, + { + "developer": "Amu", + "model_count": 2 + }, { "developer": "anakin87", "model_count": 1 @@ -1496,8 +224,8 @@ "model_count": 12 }, { - "developer": "anthropic", - "model_count": 21 + "developer": "Anthropic", + "model_count": 32 }, { "developer": "apple", @@ -1531,10 +259,26 @@ "developer": "ark", "model_count": 1 }, + { + "developer": "ArliAI", + "model_count": 2 + }, { "developer": "arshiaafshani", "model_count": 1 }, + { + "developer": "Arthur-LAGACHERIE", + "model_count": 1 + }, + { + "developer": "Artples", + "model_count": 2 + }, + { + "developer": "Aryanne", + "model_count": 3 + }, { "developer": "asharsha30", "model_count": 1 @@ -1547,29 +291,65 @@ "developer": "assskelad", "model_count": 1 }, + { + "developer": "AtAndDev", + "model_count": 1 + }, + { + "developer": "Ateron", + "model_count": 3 + }, { "developer": "athirdpath", "model_count": 1 }, { - "developer": "automerger", + "developer": "AtlaAI", + "model_count": 2 + }, + { + "developer": "AuraIndustries", + "model_count": 4 + }, + { + "developer": "Aurel9", + "model_count": 1 + }, + { + "developer": "automerger", + "model_count": 1 + }, + { + "developer": "avemio", + "model_count": 1 + }, + { + "developer": "awnr", + "model_count": 5 + }, + { + "developer": "aws-prototyping", + "model_count": 1 + }, + { + "developer": "axolotl-ai-co", "model_count": 1 }, { - "developer": "avemio", + "developer": "Ayush-Singh", "model_count": 1 }, { - "developer": "awnr", - "model_count": 5 + "developer": "Azure99", + "model_count": 6 }, { - "developer": "aws-prototyping", + "developer": "Ba2han", "model_count": 1 }, { - "developer": "axolotl-ai-co", - "model_count": 1 + "developer": "BAAI", + "model_count": 14 }, { "developer": "baconnier", @@ -1583,10 +363,22 @@ "developer": "bamec66557", "model_count": 27 }, + { + "developer": "Baptiste-HUVELLE-10", + "model_count": 1 + }, + { + "developer": "BEE-spoke-data", + "model_count": 9 + }, { "developer": "belztjti", "model_count": 2 }, + { + "developer": "BenevolenceMessiah", + "model_count": 2 + }, { "developer": "benhaotang", "model_count": 1 @@ -1619,10 +411,22 @@ "developer": "bigscience", "model_count": 7 }, + { + "developer": "BlackBeenie", + "model_count": 9 + }, + { + "developer": "Bllossom", + "model_count": 1 + }, { "developer": "bluuwhale", "model_count": 1 }, + { + "developer": "BoltMonkey", + "model_count": 3 + }, { "developer": "bond005", "model_count": 1 @@ -1635,10 +439,22 @@ "developer": "braindao", "model_count": 17 }, + { + "developer": "BrainWave-ML", + "model_count": 1 + }, + { + "developer": "BramVanroy", + "model_count": 4 + }, { "developer": "brgx53", "model_count": 6 }, + { + "developer": "BSC-LT", + "model_count": 2 + }, { "developer": "bunnycore", "model_count": 85 @@ -1647,18 +463,34 @@ "developer": "byroneverson", "model_count": 3 }, + { + "developer": "ByteDance", + "model_count": 1 + }, { "developer": "c10x", "model_count": 2 }, + { + "developer": "CarrotAI", + "model_count": 2 + }, { "developer": "carsenk", "model_count": 2 }, + { + "developer": "Casual-Autopsy", + "model_count": 1 + }, { "developer": "cat-searcher", "model_count": 2 }, + { + "developer": "CausalLM", + "model_count": 3 + }, { "developer": "cckm", "model_count": 1 @@ -1667,6 +499,10 @@ "developer": "cgato", "model_count": 1 }, + { + "developer": "Changgil", + "model_count": 2 + }, { "developer": "chargoddard", "model_count": 1 @@ -1675,10 +511,18 @@ "developer": "chujiezheng", "model_count": 2 }, + { + "developer": "CIR-AMS", + "model_count": 1 + }, { "developer": "cjvt", "model_count": 1 }, + { + "developer": "ClaudioItaly", + "model_count": 4 + }, { "developer": "cloudyu", "model_count": 7 @@ -1695,14 +539,54 @@ "developer": "cohere", "model_count": 15 }, + { + "developer": "CohereForAI", + "model_count": 8 + }, { "developer": "collaiborateorg", "model_count": 1 }, + { + "developer": "Columbia-NLP", + "model_count": 6 + }, + { + "developer": "CombinHorizon", + "model_count": 6 + }, + { + "developer": "ContactDoctor", + "model_count": 2 + }, + { + "developer": "ContextualAI", + "model_count": 16 + }, + { + "developer": "CoolSpring", + "model_count": 3 + }, + { + "developer": "Corianas", + "model_count": 3 + }, + { + "developer": "CortexLM", + "model_count": 1 + }, { "developer": "cpayne1303", "model_count": 4 }, + { + "developer": "Cran-May", + "model_count": 7 + }, + { + "developer": "CreitinGameplays", + "model_count": 1 + }, { "developer": "crestf411", "model_count": 1 @@ -1711,30 +595,98 @@ "developer": "cstr", "model_count": 1 }, + { + "developer": "CultriX", + "model_count": 32 + }, { "developer": "cyberagent", "model_count": 1 }, + { + "developer": "CYFRAGOVPL", + "model_count": 6 + }, + { + "developer": "Daemontatox", + "model_count": 32 + }, + { + "developer": "Dampfinchen", + "model_count": 1 + }, + { + "developer": "Danielbrdz", + "model_count": 7 + }, + { + "developer": "Dans-DiscountModels", + "model_count": 9 + }, { "developer": "darkc0de", "model_count": 3 }, + { + "developer": "Darkknight535", + "model_count": 1 + }, { "developer": "databricks", "model_count": 6 }, + { + "developer": "Databricks-Mosaic-Research", + "model_count": 1 + }, + { + "developer": "DavidAU", + "model_count": 25 + }, { "developer": "davidkim205", "model_count": 2 }, { - "developer": "deepseek", + "developer": "Davidsv", + "model_count": 1 + }, + { + "developer": "DavieLion", + "model_count": 5 + }, + { + "developer": "DebateLabKIT", + "model_count": 1 + }, + { + "developer": "Deci", "model_count": 2 }, + { + "developer": "DeepAutoAI", + "model_count": 12 + }, + { + "developer": "DeepMount00", + "model_count": 14 + }, + { + "developer": "DeepSeek", + "model_count": 7 + }, { "developer": "deepseek-ai", "model_count": 13 }, + { + "developer": "Delta-Vector", + "model_count": 7 + }, + { + "developer": "DevQuasar", + "model_count": 1 + }, { "developer": "dfurman", "model_count": 4 @@ -1763,10 +715,30 @@ "developer": "dnhkng", "model_count": 10 }, + { + "developer": "Dongwei", + "model_count": 1 + }, + { + "developer": "DoppelReflEx", + "model_count": 29 + }, + { + "developer": "DreadPoor", + "model_count": 119 + }, { "developer": "dreamgen", "model_count": 1 }, + { + "developer": "DRXD1000", + "model_count": 2 + }, + { + "developer": "DUAL-GPO", + "model_count": 1 + }, { "developer": "dustinwloring1988", "model_count": 7 @@ -1783,25 +755,73 @@ "developer": "dzakwan", "model_count": 1 }, + { + "developer": "DZgas", + "model_count": 1 + }, + { + "developer": "ECE-ILAB-PRYMMAL", + "model_count": 1 + }, + { + "developer": "Edgerunners", + "model_count": 1 + }, { "developer": "ehristoforu", "model_count": 36 }, { - "developer": "eleutherai", - "model_count": 2 + "developer": "EleutherAI", + "model_count": 14 + }, + { + "developer": "elinas", + "model_count": 1 + }, + { + "developer": "ell44ot", + "model_count": 1 + }, + { + "developer": "Enno-Ai", + "model_count": 4 + }, + { + "developer": "EnnoAi", + "model_count": 2 + }, + { + "developer": "Epiculous", + "model_count": 4 + }, + { + "developer": "EpistemeAI", + "model_count": 47 + }, + { + "developer": "EpistemeAI2", + "model_count": 15 + }, + { + "developer": "Eric111", + "model_count": 2 + }, + { + "developer": "Etherll", + "model_count": 8 }, { - "developer": "elinas", + "developer": "euclaise", "model_count": 1 }, { - "developer": "ell44ot", + "developer": "Eurdem", "model_count": 1 }, { - "developer": "euclaise", - "model_count": 1 + "developer": "EVA-UNIT-01", + "model_count": 2 }, { "developer": "eworojoshua", @@ -1823,18 +843,34 @@ "developer": "failspy", "model_count": 6 }, + { + "developer": "FallenMerick", + "model_count": 1 + }, { "developer": "fblgit", "model_count": 11 }, + { + "developer": "Felladrin", + "model_count": 2 + }, { "developer": "fhai50032", "model_count": 2 }, + { + "developer": "FINGU-AI", + "model_count": 7 + }, { "developer": "flammenai", "model_count": 6 }, + { + "developer": "FlofloB", + "model_count": 27 + }, { "developer": "fluently-lm", "model_count": 3 @@ -1855,14 +891,26 @@ "developer": "freewheelin", "model_count": 4 }, + { + "developer": "FuJhen", + "model_count": 4 + }, { "developer": "fulim", "model_count": 1 }, + { + "developer": "FuseAI", + "model_count": 4 + }, { "developer": "gabrielmbmb", "model_count": 1 }, + { + "developer": "GalrionSoftworks", + "model_count": 2 + }, { "developer": "gaverfraxz", "model_count": 2 @@ -1875,6 +923,10 @@ "developer": "general-preference", "model_count": 2 }, + { + "developer": "GenVRadmin", + "model_count": 4 + }, { "developer": "ghost-x", "model_count": 1 @@ -1892,8 +944,16 @@ "model_count": 26 }, { - "developer": "google", - "model_count": 64 + "developer": "Goekdeniz-Guelmez", + "model_count": 10 + }, + { + "developer": "Google", + "model_count": 68 + }, + { + "developer": "GoToCompany", + "model_count": 2 }, { "developer": "goulue5", @@ -1903,10 +963,34 @@ "developer": "gradientai", "model_count": 1 }, + { + "developer": "GreenNode", + "model_count": 1 + }, { "developer": "grimjim", "model_count": 25 }, + { + "developer": "GritLM", + "model_count": 2 + }, + { + "developer": "Groq", + "model_count": 1 + }, + { + "developer": "Gryphe", + "model_count": 5 + }, + { + "developer": "GuilhermeNaturaUmana", + "model_count": 1 + }, + { + "developer": "Gunulhona", + "model_count": 2 + }, { "developer": "gupta-tanish", "model_count": 1 @@ -1923,14 +1007,42 @@ "developer": "haoranxu", "model_count": 3 }, + { + "developer": "HarbingerX", + "model_count": 2 + }, + { + "developer": "Hastagaras", + "model_count": 3 + }, { "developer": "hatemmahmoud", "model_count": 1 }, + { + "developer": "HelpingAI", + "model_count": 4 + }, { "developer": "hendrydong", "model_count": 1 }, + { + "developer": "HeraiHench", + "model_count": 4 + }, + { + "developer": "HFXM", + "model_count": 1 + }, + { + "developer": "HiroseKoichi", + "model_count": 1 + }, + { + "developer": "HoangHa", + "model_count": 1 + }, { "developer": "hon9kon9ize", "model_count": 2 @@ -1944,24 +1056,32 @@ "model_count": 34 }, { - "developer": "huggyllama", + "developer": "HPAI-BSC", "model_count": 3 }, { - "developer": "huihui-ai", - "model_count": 8 + "developer": "HuggingFaceH4", + "model_count": 5 }, { - "developer": "huu-ontocord", - "model_count": 1 + "developer": "HuggingFaceTB", + "model_count": 12 }, { - "developer": "iFaz", + "developer": "huggyllama", + "model_count": 3 + }, + { + "developer": "huihui-ai", "model_count": 8 }, { - "developer": "iRyanBell", - "model_count": 2 + "developer": "HumanLLMs", + "model_count": 3 + }, + { + "developer": "huu-ontocord", + "model_count": 1 }, { "developer": "ibivibiv", @@ -1979,14 +1099,30 @@ "developer": "icefog72", "model_count": 62 }, + { + "developer": "IDEA-CCNL", + "model_count": 2 + }, { "developer": "ifable", "model_count": 1 }, + { + "developer": "iFaz", + "model_count": 8 + }, { "developer": "ilsp", "model_count": 1 }, + { + "developer": "IlyaGusev", + "model_count": 2 + }, + { + "developer": "Infinirc", + "model_count": 1 + }, { "developer": "inflatebot", "model_count": 1 @@ -1999,6 +1135,10 @@ "developer": "informatiker", "model_count": 1 }, + { + "developer": "INSAIT-Institute", + "model_count": 1 + }, { "developer": "insightfactory", "model_count": 1 @@ -2007,6 +1147,10 @@ "developer": "instruction-pretrain", "model_count": 1 }, + { + "developer": "Intel", + "model_count": 4 + }, { "developer": "internlm", "model_count": 9 @@ -2015,6 +1159,10 @@ "developer": "intervitens", "model_count": 1 }, + { + "developer": "IntervitensInc", + "model_count": 1 + }, { "developer": "inumulaisk", "model_count": 1 @@ -2023,6 +1171,10 @@ "developer": "invalid-coder", "model_count": 1 }, + { + "developer": "Invalid-Null", + "model_count": 2 + }, { "developer": "invisietch", "model_count": 4 @@ -2031,6 +1183,26 @@ "developer": "irahulpandey", "model_count": 1 }, + { + "developer": "iRyanBell", + "model_count": 2 + }, + { + "developer": "Isaak-Carter", + "model_count": 3 + }, + { + "developer": "J-LAB", + "model_count": 1 + }, + { + "developer": "JackFram", + "model_count": 2 + }, + { + "developer": "Jacoby746", + "model_count": 7 + }, { "developer": "jaredjoss", "model_count": 1 @@ -2043,6 +1215,10 @@ "developer": "jayasuryajsk", "model_count": 1 }, + { + "developer": "JayHyeon", + "model_count": 174 + }, { "developer": "jeanmichela", "model_count": 1 @@ -2071,6 +1247,10 @@ "developer": "jieliu", "model_count": 1 }, + { + "developer": "Jimmy19991222", + "model_count": 8 + }, { "developer": "jiviai", "model_count": 1 @@ -2087,6 +1267,14 @@ "developer": "jondurbin", "model_count": 1 }, + { + "developer": "Joseph717171", + "model_count": 2 + }, + { + "developer": "Josephgflowers", + "model_count": 7 + }, { "developer": "jpacifico", "model_count": 18 @@ -2095,6 +1283,14 @@ "developer": "jsfs11", "model_count": 3 }, + { + "developer": "JungZoona", + "model_count": 2 + }, + { + "developer": "Junhoee", + "model_count": 1 + }, { "developer": "kaist-ai", "model_count": 4 @@ -2119,6 +1315,10 @@ "developer": "kevin009", "model_count": 1 }, + { + "developer": "Khetterman", + "model_count": 2 + }, { "developer": "khoantap", "model_count": 9 @@ -2127,6 +1327,18 @@ "developer": "khulaifi95", "model_count": 1 }, + { + "developer": "Kimargin", + "model_count": 1 + }, + { + "developer": "Kimi", + "model_count": 1 + }, + { + "developer": "KingNish", + "model_count": 7 + }, { "developer": "kms7530", "model_count": 4 @@ -2135,6 +1347,30 @@ "developer": "kno10", "model_count": 2 }, + { + "developer": "Kquant03", + "model_count": 2 + }, + { + "developer": "Krystalan", + "model_count": 2 + }, + { + "developer": "KSU-HW-SEC", + "model_count": 4 + }, + { + "developer": "Kuaishou", + "model_count": 1 + }, + { + "developer": "Kukedlc", + "model_count": 7 + }, + { + "developer": "Kumar955", + "model_count": 1 + }, { "developer": "kyutai", "model_count": 1 @@ -2143,17 +1379,29 @@ "developer": "kz919", "model_count": 1 }, + { + "developer": "L-RAGE", + "model_count": 1 + }, { "developer": "ladydaina", "model_count": 1 }, { - "developer": "laislemke", + "developer": "laislemke", + "model_count": 1 + }, + { + "developer": "lalainy", + "model_count": 7 + }, + { + "developer": "Lambent", "model_count": 1 }, { - "developer": "lalainy", - "model_count": 7 + "developer": "Langboat", + "model_count": 1 }, { "developer": "langgptai", @@ -2163,22 +1411,58 @@ "developer": "lars1234", "model_count": 1 }, + { + "developer": "Lawnakk", + "model_count": 10 + }, { "developer": "leafspark", "model_count": 1 }, + { + "developer": "LEESM", + "model_count": 4 + }, { "developer": "lemon07r", "model_count": 17 }, + { + "developer": "LenguajeNaturalAI", + "model_count": 2 + }, + { + "developer": "LeroyDyer", + "model_count": 58 + }, { "developer": "lesubra", "model_count": 8 }, + { + "developer": "LGAI-EXAONE", + "model_count": 4 + }, { "developer": "lightblue", "model_count": 5 }, + { + "developer": "LightningRodLabs", + "model_count": 3 + }, + { + "developer": "Lil-R", + "model_count": 8 + }, + { + "developer": "LilRg", + "model_count": 10 + }, + { + "developer": "LimYeri", + "model_count": 5 + }, { "developer": "lkoenig", "model_count": 11 @@ -2187,6 +1471,14 @@ "developer": "llm-blender", "model_count": 1 }, + { + "developer": "LLM360", + "model_count": 2 + }, + { + "developer": "LLM4Binary", + "model_count": 1 + }, { "developer": "llmat", "model_count": 1 @@ -2199,6 +1491,10 @@ "developer": "lmsys", "model_count": 5 }, + { + "developer": "Locutusque", + "model_count": 6 + }, { "developer": "lodrick-the-lafted", "model_count": 1 @@ -2215,6 +1511,26 @@ "developer": "lunahr", "model_count": 2 }, + { + "developer": "Luni", + "model_count": 2 + }, + { + "developer": "Lunzima", + "model_count": 18 + }, + { + "developer": "LxzGordon", + "model_count": 2 + }, + { + "developer": "Lyte", + "model_count": 3 + }, + { + "developer": "M4-ai", + "model_count": 1 + }, { "developer": "m42-health", "model_count": 1 @@ -2227,10 +1543,22 @@ "developer": "magnifi", "model_count": 1 }, + { + "developer": "Magpie-Align", + "model_count": 8 + }, + { + "developer": "MagusCorp", + "model_count": 1 + }, { "developer": "maldv", "model_count": 7 }, + { + "developer": "ManoloPueblo", + "model_count": 3 + }, { "developer": "marcuscedricridia", "model_count": 40 @@ -2239,6 +1567,14 @@ "developer": "marin-community", "model_count": 1 }, + { + "developer": "MarinaraSpaghetti", + "model_count": 2 + }, + { + "developer": "Marsouuu", + "model_count": 8 + }, { "developer": "matouLeLoup", "model_count": 5 @@ -2251,6 +1587,10 @@ "developer": "maywell", "model_count": 1 }, + { + "developer": "MaziyarPanahi", + "model_count": 44 + }, { "developer": "meditsolutions", "model_count": 12 @@ -2268,8 +1608,12 @@ "model_count": 11 }, { - "developer": "meta", - "model_count": 23 + "developer": "MEscriva", + "model_count": 1 + }, + { + "developer": "Meta", + "model_count": 24 }, { "developer": "meta-llama", @@ -2295,6 +1639,10 @@ "developer": "migtissera", "model_count": 8 }, + { + "developer": "Minami-su", + "model_count": 5 + }, { "developer": "mindw96", "model_count": 1 @@ -2304,8 +1652,8 @@ "model_count": 1 }, { - "developer": "minimax", - "model_count": 1 + "developer": "MiniMax", + "model_count": 4 }, { "developer": "ministral", @@ -2335,6 +1683,10 @@ "developer": "mlabonne", "model_count": 14 }, + { + "developer": "MLP-KTLim", + "model_count": 1 + }, { "developer": "mlx-community", "model_count": 2 @@ -2347,6 +1699,14 @@ "developer": "mobiuslabsgmbh", "model_count": 2 }, + { + "developer": "ModelCloud", + "model_count": 1 + }, + { + "developer": "ModelSpace", + "model_count": 1 + }, { "developer": "moeru-ai", "model_count": 3 @@ -2355,10 +1715,18 @@ "developer": "monsterapi", "model_count": 2 }, + { + "developer": "MoonRide", + "model_count": 1 + }, { "developer": "moonshot", "model_count": 2 }, + { + "developer": "Moonshot AI", + "model_count": 2 + }, { "developer": "moonshotai", "model_count": 1 @@ -2371,6 +1739,10 @@ "developer": "mosama", "model_count": 1 }, + { + "developer": "Mostafa8Mehrabi", + "model_count": 1 + }, { "developer": "mrdayl", "model_count": 5 @@ -2379,22 +1751,54 @@ "developer": "mrm8488", "model_count": 2 }, + { + "developer": "MrRobotoAI", + "model_count": 2 + }, + { + "developer": "MTSAIR", + "model_count": 2 + }, { "developer": "mukaj", "model_count": 1 }, + { + "developer": "Multiple", + "model_count": 1 + }, + { + "developer": "MultivexAI", + "model_count": 5 + }, + { + "developer": "Mxode", + "model_count": 5 + }, { "developer": "my_model", "model_count": 1 }, + { + "developer": "NAPS-ai", + "model_count": 7 + }, { "developer": "natong19", "model_count": 2 }, + { + "developer": "Naveenpoliasetty", + "model_count": 1 + }, { "developer": "nazimali", "model_count": 2 }, + { + "developer": "NbAiLab", + "model_count": 2 + }, { "developer": "nbeerbower", "model_count": 51 @@ -2403,10 +1807,18 @@ "developer": "nbrahme", "model_count": 1 }, + { + "developer": "NCSOFT", + "model_count": 3 + }, { "developer": "necva", "model_count": 2 }, + { + "developer": "Nekochu", + "model_count": 4 + }, { "developer": "neopolita", "model_count": 11 @@ -2419,10 +1831,22 @@ "developer": "netease-youdao", "model_count": 1 }, + { + "developer": "NeverSleep", + "model_count": 2 + }, { "developer": "newsbang", "model_count": 7 }, + { + "developer": "Nexesenex", + "model_count": 45 + }, + { + "developer": "Nexusflow", + "model_count": 2 + }, { "developer": "nguyentd", "model_count": 1 @@ -2436,51 +1860,123 @@ "model_count": 5 }, { - "developer": "nicolinho", - "model_count": 4 + "developer": "nicolinho", + "model_count": 4 + }, + { + "developer": "nidum", + "model_count": 1 + }, + { + "developer": "NikolaSigmoid", + "model_count": 7 + }, + { + "developer": "nisten", + "model_count": 2 + }, + { + "developer": "Nitral-AI", + "model_count": 8 + }, + { + "developer": "NJS26", + "model_count": 1 + }, + { + "developer": "NLPark", + "model_count": 3 + }, + { + "developer": "nlpguy", + "model_count": 9 + }, + { + "developer": "Nohobby", + "model_count": 2 + }, + { + "developer": "noname0202", + "model_count": 8 + }, + { + "developer": "Norquinal", + "model_count": 8 + }, + { + "developer": "NotASI", + "model_count": 4 + }, + { + "developer": "notbdq", + "model_count": 1 + }, + { + "developer": "nothingiisreal", + "model_count": 3 + }, + { + "developer": "NousResearch", + "model_count": 19 + }, + { + "developer": "Novaciano", + "model_count": 11 + }, + { + "developer": "NTQAI", + "model_count": 2 + }, + { + "developer": "NucleusAI", + "model_count": 1 + }, + { + "developer": "nvidia", + "model_count": 21 }, { - "developer": "nidum", + "developer": "nxmwxm", "model_count": 1 }, { - "developer": "nisten", + "developer": "NYTK", "model_count": 2 }, { - "developer": "nlpguy", - "model_count": 9 + "developer": "NyxKrage", + "model_count": 1 }, { - "developer": "noname0202", - "model_count": 8 + "developer": "occiglot", + "model_count": 1 }, { - "developer": "notbdq", + "developer": "odyssey-labs", "model_count": 1 }, { - "developer": "nothingiisreal", - "model_count": 3 + "developer": "OEvortex", + "model_count": 5 }, { - "developer": "nvidia", - "model_count": 21 + "developer": "olabs-ai", + "model_count": 1 }, { - "developer": "nxmwxm", + "developer": "OliveiraJLT", "model_count": 1 }, { - "developer": "occiglot", + "developer": "Omkar1102", "model_count": 1 }, { - "developer": "odyssey-labs", + "developer": "OmnicromsBrain", "model_count": 1 }, { - "developer": "olabs-ai", + "developer": "OnlyCheeini", "model_count": 1 }, { @@ -2503,22 +1999,34 @@ "developer": "open-neo", "model_count": 2 }, + { + "developer": "Open-Orca", + "model_count": 1 + }, { "developer": "open-thoughts", "model_count": 1 }, { - "developer": "openai", - "model_count": 46 + "developer": "OpenAI", + "model_count": 59 }, { "developer": "openai-community", "model_count": 4 }, + { + "developer": "OpenAssistant", + "model_count": 4 + }, { "developer": "openbmb", "model_count": 5 }, + { + "developer": "OpenBuddy", + "model_count": 22 + }, { "developer": "openchat", "model_count": 6 @@ -2527,10 +2035,34 @@ "developer": "opencompass", "model_count": 4 }, + { + "developer": "OpenGenerativeAI", + "model_count": 2 + }, + { + "developer": "OpenLeecher", + "model_count": 1 + }, + { + "developer": "OpenLLM-France", + "model_count": 4 + }, + { + "developer": "OpenScholar", + "model_count": 1 + }, { "developer": "orai-nlp", "model_count": 1 }, + { + "developer": "Orenguteng", + "model_count": 2 + }, + { + "developer": "Orion-zhen", + "model_count": 2 + }, { "developer": "oxyapi", "model_count": 1 @@ -2543,6 +2075,10 @@ "developer": "ozone-research", "model_count": 1 }, + { + "developer": "P0x0", + "model_count": 1 + }, { "developer": "paloalma", "model_count": 5 @@ -2551,10 +2087,18 @@ "developer": "pankajmathur", "model_count": 29 }, + { + "developer": "Parissa3", + "model_count": 1 + }, { "developer": "paulml", "model_count": 1 }, + { + "developer": "Pinkstack", + "model_count": 4 + }, { "developer": "pints-ai", "model_count": 2 @@ -2563,10 +2107,46 @@ "developer": "piotr25691", "model_count": 3 }, + { + "developer": "PJMixers", + "model_count": 1 + }, + { + "developer": "PJMixers-Dev", + "model_count": 9 + }, + { + "developer": "PKU-Alignment", + "model_count": 4 + }, + { + "developer": "PocketDoc", + "model_count": 5 + }, + { + "developer": "PoLL", + "model_count": 1 + }, { "developer": "postbot", "model_count": 1 }, + { + "developer": "PowerInfer", + "model_count": 1 + }, + { + "developer": "PranavHarshan", + "model_count": 2 + }, + { + "developer": "Pretergeek", + "model_count": 9 + }, + { + "developer": "PrimeIntellect", + "model_count": 2 + }, { "developer": "prince-canuma", "model_count": 1 @@ -2587,6 +2167,18 @@ "developer": "pszemraj", "model_count": 2 }, + { + "developer": "PuxAI", + "model_count": 1 + }, + { + "developer": "PygmalionAI", + "model_count": 1 + }, + { + "developer": "Q-bert", + "model_count": 1 + }, { "developer": "qingy2019", "model_count": 7 @@ -2600,8 +2192,20 @@ "model_count": 1 }, { - "developer": "qwen", - "model_count": 10 + "developer": "Quazim0t0", + "model_count": 70 + }, + { + "developer": "Qwen", + "model_count": 70 + }, + { + "developer": "R-I-S-E", + "model_count": 2 + }, + { + "developer": "Rakuten", + "model_count": 3 }, { "developer": "raphgg", @@ -2611,6 +2215,14 @@ "developer": "rasyosef", "model_count": 4 }, + { + "developer": "Ray2333", + "model_count": 10 + }, + { + "developer": "RDson", + "model_count": 1 + }, { "developer": "realtreetune", "model_count": 1 @@ -2627,6 +2239,18 @@ "developer": "refuelai", "model_count": 1 }, + { + "developer": "Replete-AI", + "model_count": 9 + }, + { + "developer": "RESMPDEV", + "model_count": 2 + }, + { + "developer": "RezVortex", + "model_count": 2 + }, { "developer": "rhplus0831", "model_count": 1 @@ -2643,10 +2267,22 @@ "developer": "riaz", "model_count": 1 }, + { + "developer": "RLHFlow", + "model_count": 4 + }, { "developer": "rmdhirr", "model_count": 1 }, + { + "developer": "Ro-xe", + "model_count": 4 + }, + { + "developer": "Rombo-Org", + "model_count": 1 + }, { "developer": "rombodawg", "model_count": 14 @@ -2663,6 +2299,10 @@ "developer": "rubenroy", "model_count": 3 }, + { + "developer": "RubielLabarta", + "model_count": 1 + }, { "developer": "ruizhe1217", "model_count": 1 @@ -2671,6 +2311,10 @@ "developer": "rwitz", "model_count": 1 }, + { + "developer": "RWKV", + "model_count": 1 + }, { "developer": "sabersaleh", "model_count": 7 @@ -2679,6 +2323,10 @@ "developer": "sabersalehk", "model_count": 4 }, + { + "developer": "SaisExperiments", + "model_count": 6 + }, { "developer": "saishf", "model_count": 2 @@ -2691,10 +2339,18 @@ "developer": "sakaltcommunity", "model_count": 2 }, + { + "developer": "Sakalti", + "model_count": 66 + }, { "developer": "sakhan10", "model_count": 1 }, + { + "developer": "Salesforce", + "model_count": 4 + }, { "developer": "saltlux", "model_count": 2 @@ -2703,18 +2359,38 @@ "developer": "sam-paech", "model_count": 3 }, + { + "developer": "SanjiWatsuki", + "model_count": 2 + }, + { + "developer": "Sao10K", + "model_count": 8 + }, { "developer": "sarvamai", "model_count": 1 }, + { + "developer": "Saxo", + "model_count": 11 + }, { "developer": "schnapss", "model_count": 1 }, + { + "developer": "Schrieffer", + "model_count": 1 + }, { "developer": "sci-m-wang", "model_count": 3 }, + { + "developer": "SeaLLMs", + "model_count": 3 + }, { "developer": "securin", "model_count": 1 @@ -2723,6 +2399,18 @@ "developer": "senseable", "model_count": 1 }, + { + "developer": "SenseLLM", + "model_count": 2 + }, + { + "developer": "SentientAGI", + "model_count": 2 + }, + { + "developer": "SeppeV", + "model_count": 1 + }, { "developer": "sequelbox", "model_count": 6 @@ -2731,6 +2419,10 @@ "developer": "sethuiyer", "model_count": 6 }, + { + "developer": "SF-Foundation", + "model_count": 2 + }, { "developer": "sfairXC", "model_count": 1 @@ -2739,10 +2431,18 @@ "developer": "shadowml", "model_count": 2 }, + { + "developer": "Sharathhebbar24", + "model_count": 2 + }, { "developer": "shastraai", "model_count": 1 }, + { + "developer": "ShikaiChen", + "model_count": 1 + }, { "developer": "shivam9980", "model_count": 2 @@ -2751,13 +2451,25 @@ "developer": "shivank21", "model_count": 1 }, + { + "developer": "Shreyash2010", + "model_count": 1 + }, { "developer": "shuttleai", "model_count": 1 }, { - "developer": "shyamieee", - "model_count": 1 + "developer": "shyamieee", + "model_count": 1 + }, + { + "developer": "Sicarius-Prototyping", + "model_count": 3 + }, + { + "developer": "SicariusSicariiStuff", + "model_count": 19 }, { "developer": "silma-ai", @@ -2775,10 +2487,22 @@ "developer": "skymizer", "model_count": 1 }, + { + "developer": "SkyOrbis", + "model_count": 12 + }, + { + "developer": "Skywork", + "model_count": 15 + }, { "developer": "snowflake", "model_count": 1 }, + { + "developer": "Solshine", + "model_count": 2 + }, { "developer": "someon98", "model_count": 1 @@ -2795,10 +2519,26 @@ "developer": "sophosympatheia", "model_count": 1 }, + { + "developer": "Sorawiz", + "model_count": 2 + }, + { + "developer": "Sourjayon", + "model_count": 2 + }, + { + "developer": "SpaceYL", + "model_count": 1 + }, { "developer": "speakleash", "model_count": 5 }, + { + "developer": "Spestly", + "model_count": 3 + }, { "developer": "spmurrayzzz", "model_count": 1 @@ -2823,6 +2563,18 @@ "developer": "stanfordnlp", "model_count": 2 }, + { + "developer": "Stark2008", + "model_count": 3 + }, + { + "developer": "Steelskull", + "model_count": 2 + }, + { + "developer": "StelleX", + "model_count": 2 + }, { "developer": "sthenno", "model_count": 9 @@ -2843,6 +2595,10 @@ "developer": "suayptalha", "model_count": 12 }, + { + "developer": "SultanR", + "model_count": 4 + }, { "developer": "sumink", "model_count": 22 @@ -2851,14 +2607,30 @@ "developer": "sunbaby", "model_count": 1 }, + { + "developer": "Supichi", + "model_count": 11 + }, + { + "developer": "Svak", + "model_count": 2 + }, { "developer": "swap-uniba", "model_count": 1 }, + { + "developer": "Syed-Hasan-8503", + "model_count": 1 + }, { "developer": "synergetic", "model_count": 1 }, + { + "developer": "T145", + "model_count": 51 + }, { "developer": "talha2001", "model_count": 1 @@ -2875,10 +2647,26 @@ "developer": "tannedbum", "model_count": 4 }, + { + "developer": "Tarek07", + "model_count": 2 + }, + { + "developer": "TeeZee", + "model_count": 1 + }, { "developer": "teknium", "model_count": 5 }, + { + "developer": "Telugu-LLM-Labs", + "model_count": 2 + }, + { + "developer": "TencentARC", + "model_count": 4 + }, { "developer": "tensopolis", "model_count": 15 @@ -2891,6 +2679,18 @@ "developer": "tenyx", "model_count": 1 }, + { + "developer": "TheDrummer", + "model_count": 9 + }, + { + "developer": "TheDrunkenSnail", + "model_count": 3 + }, + { + "developer": "TheHierophant", + "model_count": 1 + }, { "developer": "theo77186", "model_count": 1 @@ -2899,6 +2699,10 @@ "developer": "theprint", "model_count": 18 }, + { + "developer": "TheTsar1209", + "model_count": 7 + }, { "developer": "thinkcoder", "model_count": 1 @@ -2911,22 +2715,42 @@ "developer": "thomas-yanxin", "model_count": 4 }, + { + "developer": "THUDM", + "model_count": 5 + }, { "developer": "tianyil1", "model_count": 1 }, + { + "developer": "TIGER-Lab", + "model_count": 6 + }, { "developer": "tiiuae", "model_count": 20 }, + { + "developer": "Tijmen2", + "model_count": 1 + }, { "developer": "tinycompany", "model_count": 15 }, + { + "developer": "TinyLlama", + "model_count": 6 + }, { "developer": "tklohj", "model_count": 1 }, + { + "developer": "ToastyPigeon", + "model_count": 1 + }, { "developer": "together", "model_count": 4 @@ -2943,21 +2767,57 @@ "developer": "tomasmcm", "model_count": 1 }, + { + "developer": "Trappu", + "model_count": 2 + }, + { + "developer": "Tremontaine", + "model_count": 1 + }, + { + "developer": "Triangle104", + "model_count": 61 + }, { "developer": "trthminh1112", "model_count": 1 }, + { + "developer": "Tsunami-th", + "model_count": 4 + }, + { + "developer": "TTTXXX01", + "model_count": 1 + }, { "developer": "tugstugi", "model_count": 1 }, + { + "developer": "UCLA-AGI", + "model_count": 10 + }, + { + "developer": "UKzExecution", + "model_count": 1 + }, + { + "developer": "Unbabel", + "model_count": 1 + }, + { + "developer": "Undi95", + "model_count": 2 + }, { "developer": "universalml", "model_count": 1 }, { "developer": "unknown", - "model_count": 7 + "model_count": 10 }, { "developer": "unsloth", @@ -2979,6 +2839,18 @@ "developer": "v000000", "model_count": 6 }, + { + "developer": "V3N0M", + "model_count": 1 + }, + { + "developer": "VAGOsolutions", + "model_count": 17 + }, + { + "developer": "ValiantLabs", + "model_count": 11 + }, { "developer": "vhab10", "model_count": 3 @@ -2995,6 +2867,14 @@ "developer": "vihangd", "model_count": 1 }, + { + "developer": "Vikhrmodels", + "model_count": 2 + }, + { + "developer": "VIRNECT", + "model_count": 2 + }, { "developer": "voidful", "model_count": 1 @@ -3035,6 +2915,10 @@ "developer": "weqweasdas", "model_count": 5 }, + { + "developer": "Weyaxi", + "model_count": 8 + }, { "developer": "win10", "model_count": 9 @@ -3043,6 +2927,14 @@ "developer": "winglian", "model_count": 2 }, + { + "developer": "WizardLMTeam", + "model_count": 3 + }, + { + "developer": "Wladastic", + "model_count": 1 + }, { "developer": "writer", "model_count": 7 @@ -3057,24 +2949,32 @@ }, { "developer": "xAI", - "model_count": 2 + "model_count": 7 }, { - "developer": "xMaulana", - "model_count": 1 + "developer": "Xclbr7", + "model_count": 4 }, { - "developer": "xai", - "model_count": 5 + "developer": "Xiaojian9992024", + "model_count": 12 }, { "developer": "xinchen9", "model_count": 5 }, + { + "developer": "Xkev", + "model_count": 1 + }, { "developer": "xkp24", "model_count": 8 }, + { + "developer": "xMaulana", + "model_count": 1 + }, { "developer": "xukp20", "model_count": 8 @@ -3099,6 +2999,10 @@ "developer": "yanng1242", "model_count": 1 }, + { + "developer": "Yash21", + "model_count": 1 + }, { "developer": "yasserrmd", "model_count": 2 @@ -3123,14 +3027,42 @@ "developer": "ymcki", "model_count": 11 }, + { + "developer": "Youlln", + "model_count": 19 + }, + { + "developer": "YoungPanda", + "model_count": 1 + }, + { + "developer": "YOYO-AI", + "model_count": 21 + }, { "developer": "yuchenxie", "model_count": 2 }, + { + "developer": "Yuma42", + "model_count": 3 + }, { "developer": "yuvraj17", "model_count": 3 }, + { + "developer": "Z-AI", + "model_count": 2 + }, + { + "developer": "Z.AI", + "model_count": 2 + }, + { + "developer": "Z1-Coder", + "model_count": 1 + }, { "developer": "zai-org", "model_count": 1 @@ -3143,10 +3075,22 @@ "developer": "zelk12", "model_count": 78 }, + { + "developer": "ZeroXClem", + "model_count": 11 + }, { "developer": "zetasepic", "model_count": 2 }, + { + "developer": "ZeusLabs", + "model_count": 1 + }, + { + "developer": "ZhangShenao", + "model_count": 1 + }, { "developer": "zhengr", "model_count": 1 @@ -3158,5 +3102,13 @@ { "developer": "zhipu-ai", "model_count": 1 + }, + { + "developer": "ZHLiu627", + "model_count": 2 + }, + { + "developer": "ZiyiYe", + "model_count": 1 } ] \ No newline at end of file diff --git a/data/developers/1-800-LLMs.json b/data/developers/1-800-llms.json similarity index 100% rename from data/developers/1-800-LLMs.json rename to data/developers/1-800-llms.json diff --git a/data/developers/152334H.json b/data/developers/152334h.json similarity index 100% rename from data/developers/152334H.json rename to data/developers/152334h.json diff --git a/data/developers/1TuanPham.json b/data/developers/1tuanpham.json similarity index 100% rename from data/developers/1TuanPham.json rename to data/developers/1tuanpham.json diff --git a/data/developers/3rd-Degree-Burn.json b/data/developers/3rd-degree-burn.json similarity index 100% rename from data/developers/3rd-Degree-Burn.json rename to data/developers/3rd-degree-burn.json diff --git a/data/developers/Alibaba.json b/data/developers/Alibaba.json deleted file mode 100644 index 8a87c20f0276d3b198b12dc620a9e1f6f6cd3f5f..0000000000000000000000000000000000000000 --- a/data/developers/Alibaba.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "developer": "Alibaba", - "models": [ - { - "id": "alibaba/qwen-3-coder-480b", - "name": "Qwen 3 Coder 480B", - "developer": "Alibaba", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.9 - } - }, - { - "id": "alibaba/qwen3-235b-a22b-thinking-2507", - "name": "qwen3-235b-a22b-thinking-2507", - "developer": "Alibaba", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.1267605633802817, - "livecodebenchpro/Easy Problems": 0.7605633802816901 - } - }, - { - "id": "alibaba/qwen3-30b-a3b", - "name": "qwen3-30b-a3b", - "developer": "Alibaba", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.028169014084507043, - "livecodebenchpro/Easy Problems": 0.5774647887323944 - } - }, - { - "id": "alibaba/qwen3-max", - "name": "alibaba/qwen3-max", - "developer": "Alibaba", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.04225352112676056, - "livecodebenchpro/Easy Problems": 0.36619718309859156 - } - }, - { - "id": "alibaba/qwen3-next-80b-a3b-thinking", - "name": "qwen3-next-80b-a3b-thinking", - "developer": "Alibaba", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.14084507042253522, - "livecodebenchpro/Easy Problems": 0.7464788732394366 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Anthropic.json b/data/developers/Anthropic.json deleted file mode 100644 index 21f20c31a6454ac03058235d56cee7152af73904..0000000000000000000000000000000000000000 --- a/data/developers/Anthropic.json +++ /dev/null @@ -1,129 +0,0 @@ -{ - "developer": "Anthropic", - "models": [ - { - "id": "Anthropic/claude-3-5-sonnet-20240620", - "name": "Anthropic/claude-3-5-sonnet-20240620", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.8417, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.7401, - "reward-bench/Safety": 0.8162, - "reward-bench/Reasoning": 0.8469 - } - }, - { - "id": "Anthropic/claude-3-haiku-20240307", - "name": "Anthropic/claude-3-haiku-20240307", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.7289, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Safety": 0.7953, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635 - } - }, - { - "id": "Anthropic/claude-3-opus-20240229", - "name": "Anthropic/claude-3-opus-20240229", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.8008, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.8662, - "reward-bench/Reasoning": 0.7868 - } - }, - { - "id": "Anthropic/claude-3-sonnet-20240229", - "name": "Anthropic/claude-3-sonnet-20240229", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.7458, - "reward-bench/Chat": 0.9344, - "reward-bench/Chat Hard": 0.5658, - "reward-bench/Safety": 0.8169, - "reward-bench/Reasoning": 0.6907, - "reward-bench/Prior Sets (0.5 weight)": 0.6963 - } - }, - { - "id": "anthropic/claude-3.7-sonnet", - "name": "anthropic/claude-3.7-sonnet", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.014084507042253521, - "livecodebenchpro/Easy Problems": 0.15492957746478872 - } - }, - { - "id": "anthropic/claude-haiku-4.5", - "name": "Claude Haiku 4.5", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.5 - } - }, - { - "id": "anthropic/claude-opus-4-5", - "name": "claude-opus-4-5", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.66, - "browsecompplus/browsecompplus": 0.49, - "swe-bench/swe-bench": 0.65, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, - "tau-bench-2_retail/tau-bench-2/retail": 0.85, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.58 - } - }, - { - "id": "anthropic/claude-opus-4.1", - "name": "Claude Opus 4.1", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 38.0 - } - }, - { - "id": "anthropic/claude-opus-4.5", - "name": "Claude Opus 4.5", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 54.3 - } - }, - { - "id": "anthropic/claude-opus-4.6", - "name": "Claude Opus 4.6", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 69.9 - } - }, - { - "id": "anthropic/claude-sonnet-4.5", - "name": "Claude Sonnet 4.5", - "developer": "Anthropic", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.6 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/DeepSeek.json b/data/developers/DeepSeek.json deleted file mode 100644 index b7172f25cd90b14cf61518ba803a99d37ec6b744..0000000000000000000000000000000000000000 --- a/data/developers/DeepSeek.json +++ /dev/null @@ -1,58 +0,0 @@ -{ - "developer": "DeepSeek", - "models": [ - { - "id": "deepseek/chat-v3-0324", - "name": "deepseek/chat-v3-0324", - "developer": "DeepSeek", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.0, - "livecodebenchpro/Easy Problems": 0.19718309859154928 - } - }, - { - "id": "deepseek/deepseek-v3.2", - "name": "DeepSeek-V3.2", - "developer": "DeepSeek", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 39.6 - } - }, - { - "id": "deepseek/ep-20250214004308-p7n89", - "name": "ep-20250214004308-p7n89", - "developer": "DeepSeek", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.014084507042253521, - "livecodebenchpro/Easy Problems": 0.4225352112676056 - } - }, - { - "id": "deepseek/ep-20250228232227-z44x5", - "name": "ep-20250228232227-z44x5", - "developer": "DeepSeek", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.0, - "livecodebenchpro/Easy Problems": 0.1267605633802817 - } - }, - { - "id": "deepseek/ep-20250603132404-cgpjm", - "name": "ep-20250603132404-cgpjm", - "developer": "DeepSeek", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.08450704225352113, - "livecodebenchpro/Easy Problems": 0.5774647887323944 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/EleutherAI.json b/data/developers/EleutherAI.json deleted file mode 100644 index 3581e281f97cadb3a960db5014920f2d801e0606..0000000000000000000000000000000000000000 --- a/data/developers/EleutherAI.json +++ /dev/null @@ -1,173 +0,0 @@ -{ - "developer": "EleutherAI", - "models": [ - { - "id": "EleutherAI/gpt-j-6b", - "name": "gpt-j-6b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2522, - "hfopenllm_v2/BBH": 0.3191, - "hfopenllm_v2/MATH Level 5": 0.0136, - "hfopenllm_v2/GPQA": 0.2458, - "hfopenllm_v2/MUSR": 0.3658, - "hfopenllm_v2/MMLU-PRO": 0.1241 - } - }, - { - "id": "EleutherAI/gpt-neo-1.3B", - "name": "gpt-neo-1.3B", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2079, - "hfopenllm_v2/BBH": 0.3039, - "hfopenllm_v2/MATH Level 5": 0.0106, - "hfopenllm_v2/GPQA": 0.2559, - "hfopenllm_v2/MUSR": 0.3817, - "hfopenllm_v2/MMLU-PRO": 0.1164 - } - }, - { - "id": "EleutherAI/gpt-neo-125m", - "name": "gpt-neo-125m", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1905, - "hfopenllm_v2/BBH": 0.3115, - "hfopenllm_v2/MATH Level 5": 0.006, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3593, - "hfopenllm_v2/MMLU-PRO": 0.1026 - } - }, - { - "id": "EleutherAI/gpt-neo-2.7B", - "name": "gpt-neo-2.7B", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.259, - "hfopenllm_v2/BBH": 0.314, - "hfopenllm_v2/MATH Level 5": 0.0106, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3554, - "hfopenllm_v2/MMLU-PRO": 0.1163 - } - }, - { - "id": "EleutherAI/gpt-neox-20b", - "name": "gpt-neox-20b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2587, - "hfopenllm_v2/BBH": 0.3165, - "hfopenllm_v2/MATH Level 5": 0.0136, - "hfopenllm_v2/GPQA": 0.2433, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1155 - } - }, - { - "id": "EleutherAI/pythia-1.4b", - "name": "pythia-1.4b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2371, - "hfopenllm_v2/BBH": 0.315, - "hfopenllm_v2/MATH Level 5": 0.0151, - "hfopenllm_v2/GPQA": 0.2617, - "hfopenllm_v2/MUSR": 0.3538, - "hfopenllm_v2/MMLU-PRO": 0.1123 - } - }, - { - "id": "EleutherAI/pythia-12b", - "name": "pythia-12b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2471, - "hfopenllm_v2/BBH": 0.318, - "hfopenllm_v2/MATH Level 5": 0.0166, - "hfopenllm_v2/GPQA": 0.2466, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1109 - } - }, - { - "id": "EleutherAI/pythia-160m", - "name": "pythia-160m", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1816, - "hfopenllm_v2/BBH": 0.297, - "hfopenllm_v2/MATH Level 5": 0.0091, - "hfopenllm_v2/GPQA": 0.2584, - "hfopenllm_v2/MUSR": 0.4179, - "hfopenllm_v2/MMLU-PRO": 0.112 - } - }, - { - "id": "EleutherAI/pythia-1b", - "name": "pythia-1b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2208, - "hfopenllm_v2/BBH": 0.3004, - "hfopenllm_v2/MATH Level 5": 0.0091, - "hfopenllm_v2/GPQA": 0.2567, - "hfopenllm_v2/MUSR": 0.3552, - "hfopenllm_v2/MMLU-PRO": 0.1136 - } - }, - { - "id": "EleutherAI/pythia-2.8b", - "name": "pythia-2.8b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2173, - "hfopenllm_v2/BBH": 0.3224, - "hfopenllm_v2/MATH Level 5": 0.0136, - "hfopenllm_v2/GPQA": 0.25, - "hfopenllm_v2/MUSR": 0.3486, - "hfopenllm_v2/MMLU-PRO": 0.1137 - } - }, - { - "id": "EleutherAI/pythia-410m", - "name": "pythia-410m", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2195, - "hfopenllm_v2/BBH": 0.3028, - "hfopenllm_v2/MATH Level 5": 0.0098, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3578, - "hfopenllm_v2/MMLU-PRO": 0.1128 - } - }, - { - "id": "EleutherAI/pythia-6.9b", - "name": "pythia-6.9b", - "developer": "EleutherAI", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2281, - "hfopenllm_v2/BBH": 0.3232, - "hfopenllm_v2/MATH Level 5": 0.0144, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3591, - "hfopenllm_v2/MMLU-PRO": 0.1147 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Google.json b/data/developers/Google.json deleted file mode 100644 index a72418ac229d3b0c23d277e602231dc9424cf54c..0000000000000000000000000000000000000000 --- a/data/developers/Google.json +++ /dev/null @@ -1,65 +0,0 @@ -{ - "developer": "Google", - "models": [ - { - "id": "google/gemini-3-flash", - "name": "Gemini 3 Flash", - "developer": "Google", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 47.4 - } - }, - { - "id": "google/gemini-3-pro", - "name": "Gemini 3 Pro", - "developer": "Google", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.2 - } - }, - { - "id": "google/gemini-3-pro-preview", - "name": "gemini-3-pro-preview", - "developer": "Google", - "evaluator_relationship": null, - "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.505, - "browsecompplus/browsecompplus": 0.48, - "global-mmlu-lite/Global MMLU Lite": 0.9453, - "global-mmlu-lite/Culturally Sensitive": 0.9397, - "global-mmlu-lite/Culturally Agnostic": 0.9509, - "global-mmlu-lite/Arabic": 0.9475, - "global-mmlu-lite/English": 0.9425, - "global-mmlu-lite/Bengali": 0.9425, - "global-mmlu-lite/German": 0.94, - "global-mmlu-lite/French": 0.9575, - "global-mmlu-lite/Hindi": 0.9425, - "global-mmlu-lite/Indonesian": 0.955, - "global-mmlu-lite/Italian": 0.955, - "global-mmlu-lite/Japanese": 0.94, - "global-mmlu-lite/Korean": 0.94, - "global-mmlu-lite/Portuguese": 0.9425, - "global-mmlu-lite/Spanish": 0.9475, - "global-mmlu-lite/Swahili": 0.94, - "global-mmlu-lite/Yoruba": 0.9425, - "global-mmlu-lite/Chinese": 0.9475, - "global-mmlu-lite/Burmese": 0.9425, - "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, - "tau-bench-2_retail/tau-bench-2/retail": 0.7805, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 - } - }, - { - "id": "google/gemini-3.1-pro", - "name": "Gemini 3.1 Pro", - "developer": "Google", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 78.4 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Meta.json b/data/developers/Meta.json deleted file mode 100644 index 22227f5c93a37eaf8487a1078b002421f17c3dec..0000000000000000000000000000000000000000 --- a/data/developers/Meta.json +++ /dev/null @@ -1,16 +0,0 @@ -{ - "developer": "Meta", - "models": [ - { - "id": "meta/llama-4-maverick", - "name": "meta/llama-4-maverick", - "developer": "Meta", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.0, - "livecodebenchpro/Easy Problems": 0.09859154929577464 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/MiniMax.json b/data/developers/MiniMax.json deleted file mode 100644 index bf1e95583288fed674acdea5cb3109f4a1eaa232..0000000000000000000000000000000000000000 --- a/data/developers/MiniMax.json +++ /dev/null @@ -1,23 +0,0 @@ -{ - "developer": "MiniMax", - "models": [ - { - "id": "minimax/minimax-m2", - "name": "MiniMax M2", - "developer": "MiniMax", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 30.0 - } - }, - { - "id": "minimax/minimax-m2.1", - "name": "MiniMax M2.1", - "developer": "MiniMax", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 29.2 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Minimax.json b/data/developers/Minimax.json deleted file mode 100644 index a47e6e8c5b7af8d07d5ab11c54cde104c2cb3327..0000000000000000000000000000000000000000 --- a/data/developers/Minimax.json +++ /dev/null @@ -1,14 +0,0 @@ -{ - "developer": "Minimax", - "models": [ - { - "id": "minimax/minimax-m2.5", - "name": "Minimax m2.5", - "developer": "Minimax", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.2 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/OpenAI.json b/data/developers/OpenAI.json deleted file mode 100644 index 4f5fe50e33b6537a7fa64519c50297e6a6d53b51..0000000000000000000000000000000000000000 --- a/data/developers/OpenAI.json +++ /dev/null @@ -1,132 +0,0 @@ -{ - "developer": "OpenAI", - "models": [ - { - "id": "openai/gpt-4.1", - "name": "openai/gpt-4.1", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "livecodebenchpro/Hard Problems": 0.0, - "livecodebenchpro/Medium Problems": 0.0, - "livecodebenchpro/Easy Problems": 0.19718309859154928 - } - }, - { - "id": "openai/gpt-5", - "name": "GPT-5", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.2 - } - }, - { - "id": "openai/gpt-5-codex", - "name": "GPT-5-Codex", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 41.3 - } - }, - { - "id": "openai/gpt-5-mini", - "name": "GPT-5-Mini", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 31.9 - } - }, - { - "id": "openai/gpt-5-nano", - "name": "GPT-5-Nano", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 - } - }, - { - "id": "openai/gpt-5.1", - "name": "GPT-5.1", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 47.6 - } - }, - { - "id": "openai/gpt-5.1-codex", - "name": "GPT-5.1-Codex", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 - } - }, - { - "id": "openai/gpt-5.1-codex-max", - "name": "GPT-5.1-Codex-Max", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.4 - } - }, - { - "id": "openai/gpt-5.1-codex-mini", - "name": "GPT-5.1-Codex-Mini", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 43.1 - } - }, - { - "id": "openai/gpt-5.2", - "name": "GPT-5.2", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 60.7 - } - }, - { - "id": "openai/gpt-5.2-2025-12-11", - "name": "gpt-5.2-2025-12-11", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.071, - "browsecompplus/browsecompplus": 0.46, - "livecodebenchpro/Hard Problems": 0.1594, - "livecodebenchpro/Medium Problems": 0.5211, - "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.5253, - "tau-bench-2_airline/tau-bench-2/airline": 0.54, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 - } - }, - { - "id": "openai/gpt-5.2-codex", - "name": "GPT-5.2-Codex", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 66.5 - } - }, - { - "id": "openai/gpt-5.3-codex", - "name": "GPT-5.3-Codex", - "developer": "OpenAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 77.3 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Qwen.json b/data/developers/Qwen.json deleted file mode 100644 index 782ee6834856d0a95850e189e3f494255390e68a..0000000000000000000000000000000000000000 --- a/data/developers/Qwen.json +++ /dev/null @@ -1,882 +0,0 @@ -{ - "developer": "Qwen", - "models": [ - { - "id": "Qwen/QwQ-32B", - "name": "QwQ-32B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3977, - "hfopenllm_v2/BBH": 0.2983, - "hfopenllm_v2/MATH Level 5": 0.1609, - "hfopenllm_v2/GPQA": 0.2601, - "hfopenllm_v2/MUSR": 0.4206, - "hfopenllm_v2/MMLU-PRO": 0.1196 - } - }, - { - "id": "Qwen/QwQ-32B-Preview", - "name": "QwQ-32B-Preview", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4035, - "hfopenllm_v2/BBH": 0.6691, - "hfopenllm_v2/MATH Level 5": 0.4494, - "hfopenllm_v2/GPQA": 0.2819, - "hfopenllm_v2/MUSR": 0.411, - "hfopenllm_v2/MMLU-PRO": 0.5678 - } - }, - { - "id": "Qwen/Qwen1.5-0.5B", - "name": "Qwen1.5-0.5B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1706, - "hfopenllm_v2/BBH": 0.3154, - "hfopenllm_v2/MATH Level 5": 0.0174, - "hfopenllm_v2/GPQA": 0.2542, - "hfopenllm_v2/MUSR": 0.3616, - "hfopenllm_v2/MMLU-PRO": 0.1307 - } - }, - { - "id": "Qwen/Qwen1.5-0.5B-Chat", - "name": "Qwen1.5-0.5B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1807, - "hfopenllm_v2/BBH": 0.3167, - "hfopenllm_v2/MATH Level 5": 0.0068, - "hfopenllm_v2/GPQA": 0.2693, - "hfopenllm_v2/MUSR": 0.3837, - "hfopenllm_v2/MMLU-PRO": 0.1213, - "reward-bench/Score": 0.5298, - "reward-bench/Chat": 0.3547, - "reward-bench/Chat Hard": 0.6294, - "reward-bench/Safety": 0.5703, - "reward-bench/Reasoning": 0.5984, - "reward-bench/Prior Sets (0.5 weight)": 0.4629 - } - }, - { - "id": "Qwen/Qwen1.5-1.8B", - "name": "Qwen1.5-1.8B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2154, - "hfopenllm_v2/BBH": 0.3476, - "hfopenllm_v2/MATH Level 5": 0.0317, - "hfopenllm_v2/GPQA": 0.3054, - "hfopenllm_v2/MUSR": 0.3605, - "hfopenllm_v2/MMLU-PRO": 0.1882 - } - }, - { - "id": "Qwen/Qwen1.5-1.8B-Chat", - "name": "Qwen1.5-1.8B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2019, - "hfopenllm_v2/BBH": 0.3256, - "hfopenllm_v2/MATH Level 5": 0.0196, - "hfopenllm_v2/GPQA": 0.2978, - "hfopenllm_v2/MUSR": 0.426, - "hfopenllm_v2/MMLU-PRO": 0.1804, - "reward-bench/Score": 0.589, - "reward-bench/Chat": 0.5615, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.4838, - "reward-bench/Reasoning": 0.7793, - "reward-bench/Prior Sets (0.5 weight)": 0.4453 - } - }, - { - "id": "Qwen/Qwen1.5-110B", - "name": "Qwen1.5-110B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3422, - "hfopenllm_v2/BBH": 0.61, - "hfopenllm_v2/MATH Level 5": 0.247, - "hfopenllm_v2/GPQA": 0.3523, - "hfopenllm_v2/MUSR": 0.4408, - "hfopenllm_v2/MMLU-PRO": 0.5361 - } - }, - { - "id": "Qwen/Qwen1.5-110B-Chat", - "name": "Qwen1.5-110B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5939, - "hfopenllm_v2/BBH": 0.6184, - "hfopenllm_v2/MATH Level 5": 0.2341, - "hfopenllm_v2/GPQA": 0.3414, - "hfopenllm_v2/MUSR": 0.4522, - "hfopenllm_v2/MMLU-PRO": 0.4825 - } - }, - { - "id": "Qwen/Qwen1.5-14B", - "name": "Qwen1.5-14B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2905, - "hfopenllm_v2/BBH": 0.508, - "hfopenllm_v2/MATH Level 5": 0.2024, - "hfopenllm_v2/GPQA": 0.2945, - "hfopenllm_v2/MUSR": 0.4186, - "hfopenllm_v2/MMLU-PRO": 0.3644 - } - }, - { - "id": "Qwen/Qwen1.5-14B-Chat", - "name": "Qwen1.5-14B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4768, - "hfopenllm_v2/BBH": 0.5229, - "hfopenllm_v2/MATH Level 5": 0.1526, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.44, - "hfopenllm_v2/MMLU-PRO": 0.3618, - "reward-bench/Score": 0.6864, - "reward-bench/Chat": 0.5726, - "reward-bench/Chat Hard": 0.7018, - "reward-bench/Safety": 0.7122, - "reward-bench/Reasoning": 0.8961, - "reward-bench/Prior Sets (0.5 weight)": 0.4123 - } - }, - { - "id": "Qwen/Qwen1.5-32B", - "name": "Qwen1.5-32B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3297, - "hfopenllm_v2/BBH": 0.5715, - "hfopenllm_v2/MATH Level 5": 0.3029, - "hfopenllm_v2/GPQA": 0.3297, - "hfopenllm_v2/MUSR": 0.4278, - "hfopenllm_v2/MMLU-PRO": 0.45 - } - }, - { - "id": "Qwen/Qwen1.5-32B-Chat", - "name": "Qwen1.5-32B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5532, - "hfopenllm_v2/BBH": 0.6067, - "hfopenllm_v2/MATH Level 5": 0.1956, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.416, - "hfopenllm_v2/MMLU-PRO": 0.4457 - } - }, - { - "id": "Qwen/Qwen1.5-4B", - "name": "Qwen1.5-4B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2445, - "hfopenllm_v2/BBH": 0.4054, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.3604, - "hfopenllm_v2/MMLU-PRO": 0.246 - } - }, - { - "id": "Qwen/Qwen1.5-4B-Chat", - "name": "Qwen1.5-4B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3157, - "hfopenllm_v2/BBH": 0.4006, - "hfopenllm_v2/MATH Level 5": 0.0279, - "hfopenllm_v2/GPQA": 0.2668, - "hfopenllm_v2/MUSR": 0.3978, - "hfopenllm_v2/MMLU-PRO": 0.2396, - "reward-bench/Score": 0.5477, - "reward-bench/Chat": 0.3883, - "reward-bench/Chat Hard": 0.6272, - "reward-bench/Safety": 0.5568, - "reward-bench/Reasoning": 0.6689, - "reward-bench/Prior Sets (0.5 weight)": 0.447 - } - }, - { - "id": "Qwen/Qwen1.5-72B-Chat", - "name": "Qwen/Qwen1.5-72B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.6723, - "reward-bench/Chat": 0.6229, - "reward-bench/Chat Hard": 0.6601, - "reward-bench/Safety": 0.6757, - "reward-bench/Reasoning": 0.8554, - "reward-bench/Prior Sets (0.5 weight)": 0.4226 - } - }, - { - "id": "Qwen/Qwen1.5-7B", - "name": "Qwen1.5-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2684, - "hfopenllm_v2/BBH": 0.456, - "hfopenllm_v2/MATH Level 5": 0.0929, - "hfopenllm_v2/GPQA": 0.2987, - "hfopenllm_v2/MUSR": 0.4103, - "hfopenllm_v2/MMLU-PRO": 0.2916 - } - }, - { - "id": "Qwen/Qwen1.5-7B-Chat", - "name": "Qwen1.5-7B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4371, - "hfopenllm_v2/BBH": 0.451, - "hfopenllm_v2/MATH Level 5": 0.0627, - "hfopenllm_v2/GPQA": 0.3029, - "hfopenllm_v2/MUSR": 0.3779, - "hfopenllm_v2/MMLU-PRO": 0.2951, - "reward-bench/Score": 0.675, - "reward-bench/Chat": 0.5363, - "reward-bench/Chat Hard": 0.6908, - "reward-bench/Safety": 0.6919, - "reward-bench/Reasoning": 0.9041, - "reward-bench/Prior Sets (0.5 weight)": 0.4288 - } - }, - { - "id": "Qwen/Qwen1.5-MoE-A2.7B", - "name": "Qwen1.5-MoE-A2.7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.266, - "hfopenllm_v2/BBH": 0.4114, - "hfopenllm_v2/MATH Level 5": 0.0929, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.4013, - "hfopenllm_v2/MMLU-PRO": 0.2778 - } - }, - { - "id": "Qwen/Qwen1.5-MoE-A2.7B-Chat", - "name": "Qwen1.5-MoE-A2.7B-Chat", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3795, - "hfopenllm_v2/BBH": 0.4272, - "hfopenllm_v2/MATH Level 5": 0.0634, - "hfopenllm_v2/GPQA": 0.2743, - "hfopenllm_v2/MUSR": 0.3899, - "hfopenllm_v2/MMLU-PRO": 0.2923, - "reward-bench/Score": 0.6644, - "reward-bench/Chat": 0.7291, - "reward-bench/Chat Hard": 0.6316, - "reward-bench/Safety": 0.6284, - "reward-bench/Reasoning": 0.774, - "reward-bench/Prior Sets (0.5 weight)": 0.4536 - } - }, - { - "id": "Qwen/Qwen2-0.5B", - "name": "Qwen2-0.5B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1873, - "hfopenllm_v2/BBH": 0.3239, - "hfopenllm_v2/MATH Level 5": 0.0264, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3752, - "hfopenllm_v2/MMLU-PRO": 0.172 - } - }, - { - "id": "Qwen/Qwen2-0.5B-Instruct", - "name": "Qwen2-0.5B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2247, - "hfopenllm_v2/BBH": 0.3173, - "hfopenllm_v2/MATH Level 5": 0.0287, - "hfopenllm_v2/GPQA": 0.2466, - "hfopenllm_v2/MUSR": 0.3353, - "hfopenllm_v2/MMLU-PRO": 0.1531 - } - }, - { - "id": "Qwen/Qwen2-1.5B", - "name": "Qwen2-1.5B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2113, - "hfopenllm_v2/BBH": 0.3575, - "hfopenllm_v2/MATH Level 5": 0.0702, - "hfopenllm_v2/GPQA": 0.2643, - "hfopenllm_v2/MUSR": 0.3658, - "hfopenllm_v2/MMLU-PRO": 0.2552 - } - }, - { - "id": "Qwen/Qwen2-1.5B-Instruct", - "name": "Qwen2-1.5B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3371, - "hfopenllm_v2/BBH": 0.3852, - "hfopenllm_v2/MATH Level 5": 0.0718, - "hfopenllm_v2/GPQA": 0.2617, - "hfopenllm_v2/MUSR": 0.4293, - "hfopenllm_v2/MMLU-PRO": 0.2501 - } - }, - { - "id": "Qwen/Qwen2-57B-A14B", - "name": "Qwen2-57B-A14B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3113, - "hfopenllm_v2/BBH": 0.5618, - "hfopenllm_v2/MATH Level 5": 0.1866, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4174, - "hfopenllm_v2/MMLU-PRO": 0.4916 - } - }, - { - "id": "Qwen/Qwen2-57B-A14B-Instruct", - "name": "Qwen2-57B-A14B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6338, - "hfopenllm_v2/BBH": 0.5888, - "hfopenllm_v2/MATH Level 5": 0.2817, - "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4361, - "hfopenllm_v2/MMLU-PRO": 0.4575 - } - }, - { - "id": "Qwen/Qwen2-72B", - "name": "Qwen2-72B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3824, - "hfopenllm_v2/BBH": 0.6617, - "hfopenllm_v2/MATH Level 5": 0.3112, - "hfopenllm_v2/GPQA": 0.3943, - "hfopenllm_v2/MUSR": 0.4704, - "hfopenllm_v2/MMLU-PRO": 0.5731 - } - }, - { - "id": "Qwen/Qwen2-72B-Instruct", - "name": "Qwen2-72B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7989, - "hfopenllm_v2/BBH": 0.6977, - "hfopenllm_v2/MATH Level 5": 0.4177, - "hfopenllm_v2/GPQA": 0.3725, - "hfopenllm_v2/MUSR": 0.456, - "hfopenllm_v2/MMLU-PRO": 0.5403 - } - }, - { - "id": "Qwen/Qwen2-7B", - "name": "Qwen2-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3149, - "hfopenllm_v2/BBH": 0.5315, - "hfopenllm_v2/MATH Level 5": 0.2039, - "hfopenllm_v2/GPQA": 0.3045, - "hfopenllm_v2/MUSR": 0.4439, - "hfopenllm_v2/MMLU-PRO": 0.4183 - } - }, - { - "id": "Qwen/Qwen2-7B-Instruct", - "name": "Qwen2-7B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5679, - "hfopenllm_v2/BBH": 0.5545, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.2978, - "hfopenllm_v2/MUSR": 0.3928, - "hfopenllm_v2/MMLU-PRO": 0.3847 - } - }, - { - "id": "Qwen/Qwen2-Math-72B-Instruct", - "name": "Qwen2-Math-72B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5694, - "hfopenllm_v2/BBH": 0.6343, - "hfopenllm_v2/MATH Level 5": 0.5536, - "hfopenllm_v2/GPQA": 0.3683, - "hfopenllm_v2/MUSR": 0.4517, - "hfopenllm_v2/MMLU-PRO": 0.4273 - } - }, - { - "id": "Qwen/Qwen2-Math-7B", - "name": "Qwen2-Math-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2687, - "hfopenllm_v2/BBH": 0.387, - "hfopenllm_v2/MATH Level 5": 0.2477, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3593, - "hfopenllm_v2/MMLU-PRO": 0.1197 - } - }, - { - "id": "Qwen/Qwen2-VL-72B-Instruct", - "name": "Qwen2-VL-72B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5982, - "hfopenllm_v2/BBH": 0.6946, - "hfopenllm_v2/MATH Level 5": 0.3444, - "hfopenllm_v2/GPQA": 0.3876, - "hfopenllm_v2/MUSR": 0.4492, - "hfopenllm_v2/MMLU-PRO": 0.5717 - } - }, - { - "id": "Qwen/Qwen2-VL-7B-Instruct", - "name": "Qwen2-VL-7B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4599, - "hfopenllm_v2/BBH": 0.5465, - "hfopenllm_v2/MATH Level 5": 0.1986, - "hfopenllm_v2/GPQA": 0.3196, - "hfopenllm_v2/MUSR": 0.4375, - "hfopenllm_v2/MMLU-PRO": 0.4095 - } - }, - { - "id": "Qwen/Qwen2.5-0.5B", - "name": "Qwen2.5-0.5B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1627, - "hfopenllm_v2/BBH": 0.3275, - "hfopenllm_v2/MATH Level 5": 0.0393, - "hfopenllm_v2/GPQA": 0.2466, - "hfopenllm_v2/MUSR": 0.3433, - "hfopenllm_v2/MMLU-PRO": 0.1906 - } - }, - { - "id": "Qwen/Qwen2.5-0.5B-Instruct", - "name": "Qwen2.5-0.5B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3071, - "hfopenllm_v2/BBH": 0.3341, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3329, - "hfopenllm_v2/MMLU-PRO": 0.1697 - } - }, - { - "id": "Qwen/Qwen2.5-1.5B", - "name": "Qwen2.5-1.5B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2674, - "hfopenllm_v2/BBH": 0.4078, - "hfopenllm_v2/MATH Level 5": 0.0914, - "hfopenllm_v2/GPQA": 0.2852, - "hfopenllm_v2/MUSR": 0.3576, - "hfopenllm_v2/MMLU-PRO": 0.2855 - } - }, - { - "id": "Qwen/Qwen2.5-1.5B-Instruct", - "name": "Qwen2.5-1.5B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4476, - "hfopenllm_v2/BBH": 0.4289, - "hfopenllm_v2/MATH Level 5": 0.2205, - "hfopenllm_v2/GPQA": 0.2559, - "hfopenllm_v2/MUSR": 0.3663, - "hfopenllm_v2/MMLU-PRO": 0.2799 - } - }, - { - "id": "Qwen/Qwen2.5-14B", - "name": "Qwen2.5-14B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3694, - "hfopenllm_v2/BBH": 0.6161, - "hfopenllm_v2/MATH Level 5": 0.29, - "hfopenllm_v2/GPQA": 0.3817, - "hfopenllm_v2/MUSR": 0.4502, - "hfopenllm_v2/MMLU-PRO": 0.5249 - } - }, - { - "id": "Qwen/Qwen2.5-14B-Instruct", - "name": "Qwen2.5-14B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8158, - "hfopenllm_v2/BBH": 0.639, - "hfopenllm_v2/MATH Level 5": 0.5476, - "hfopenllm_v2/GPQA": 0.3221, - "hfopenllm_v2/MUSR": 0.4101, - "hfopenllm_v2/MMLU-PRO": 0.4904 - } - }, - { - "id": "Qwen/Qwen2.5-14B-Instruct-1M", - "name": "Qwen2.5-14B-Instruct-1M", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8414, - "hfopenllm_v2/BBH": 0.6198, - "hfopenllm_v2/MATH Level 5": 0.5302, - "hfopenllm_v2/GPQA": 0.3431, - "hfopenllm_v2/MUSR": 0.418, - "hfopenllm_v2/MMLU-PRO": 0.485 - } - }, - { - "id": "Qwen/Qwen2.5-32B", - "name": "Qwen2.5-32B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4077, - "hfopenllm_v2/BBH": 0.6771, - "hfopenllm_v2/MATH Level 5": 0.3565, - "hfopenllm_v2/GPQA": 0.4119, - "hfopenllm_v2/MUSR": 0.4978, - "hfopenllm_v2/MMLU-PRO": 0.5805 - } - }, - { - "id": "Qwen/Qwen2.5-32B-Instruct", - "name": "Qwen2.5-32B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8346, - "hfopenllm_v2/BBH": 0.6913, - "hfopenllm_v2/MATH Level 5": 0.6254, - "hfopenllm_v2/GPQA": 0.3381, - "hfopenllm_v2/MUSR": 0.4261, - "hfopenllm_v2/MMLU-PRO": 0.5667 - } - }, - { - "id": "Qwen/Qwen2.5-3B", - "name": "Qwen2.5-3B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.269, - "hfopenllm_v2/BBH": 0.4612, - "hfopenllm_v2/MATH Level 5": 0.148, - "hfopenllm_v2/GPQA": 0.2978, - "hfopenllm_v2/MUSR": 0.4303, - "hfopenllm_v2/MMLU-PRO": 0.3203 - } - }, - { - "id": "Qwen/Qwen2.5-3B-Instruct", - "name": "Qwen2.5-3B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6475, - "hfopenllm_v2/BBH": 0.4693, - "hfopenllm_v2/MATH Level 5": 0.3678, - "hfopenllm_v2/GPQA": 0.2727, - "hfopenllm_v2/MUSR": 0.3968, - "hfopenllm_v2/MMLU-PRO": 0.3255 - } - }, - { - "id": "Qwen/Qwen2.5-72B", - "name": "Qwen2.5-72B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4137, - "hfopenllm_v2/BBH": 0.6797, - "hfopenllm_v2/MATH Level 5": 0.3912, - "hfopenllm_v2/GPQA": 0.4052, - "hfopenllm_v2/MUSR": 0.4771, - "hfopenllm_v2/MMLU-PRO": 0.5968 - } - }, - { - "id": "Qwen/Qwen2.5-72B-Instruct", - "name": "Qwen2.5-72B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8638, - "hfopenllm_v2/BBH": 0.7273, - "hfopenllm_v2/MATH Level 5": 0.5982, - "hfopenllm_v2/GPQA": 0.375, - "hfopenllm_v2/MUSR": 0.4206, - "hfopenllm_v2/MMLU-PRO": 0.5626 - } - }, - { - "id": "Qwen/Qwen2.5-7B", - "name": "Qwen2.5-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3374, - "hfopenllm_v2/BBH": 0.5416, - "hfopenllm_v2/MATH Level 5": 0.2508, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.4424, - "hfopenllm_v2/MMLU-PRO": 0.4365 - } - }, - { - "id": "Qwen/Qwen2.5-7B-Instruct", - "name": "Qwen2.5-7B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7585, - "hfopenllm_v2/BBH": 0.5394, - "hfopenllm_v2/MATH Level 5": 0.5, - "hfopenllm_v2/GPQA": 0.2911, - "hfopenllm_v2/MUSR": 0.402, - "hfopenllm_v2/MMLU-PRO": 0.4287 - } - }, - { - "id": "Qwen/Qwen2.5-7B-Instruct-1M", - "name": "Qwen2.5-7B-Instruct-1M", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7448, - "hfopenllm_v2/BBH": 0.5404, - "hfopenllm_v2/MATH Level 5": 0.4335, - "hfopenllm_v2/GPQA": 0.2978, - "hfopenllm_v2/MUSR": 0.4087, - "hfopenllm_v2/MMLU-PRO": 0.3505 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-14B", - "name": "Qwen2.5-Coder-14B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3473, - "hfopenllm_v2/BBH": 0.5865, - "hfopenllm_v2/MATH Level 5": 0.2251, - "hfopenllm_v2/GPQA": 0.2928, - "hfopenllm_v2/MUSR": 0.3874, - "hfopenllm_v2/MMLU-PRO": 0.4521 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-14B-Instruct", - "name": "Qwen2.5-Coder-14B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6908, - "hfopenllm_v2/BBH": 0.614, - "hfopenllm_v2/MATH Level 5": 0.3248, - "hfopenllm_v2/GPQA": 0.3045, - "hfopenllm_v2/MUSR": 0.3915, - "hfopenllm_v2/MMLU-PRO": 0.3939 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-32B", - "name": "Qwen2.5-Coder-32B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4363, - "hfopenllm_v2/BBH": 0.6404, - "hfopenllm_v2/MATH Level 5": 0.3089, - "hfopenllm_v2/GPQA": 0.3465, - "hfopenllm_v2/MUSR": 0.4528, - "hfopenllm_v2/MMLU-PRO": 0.5303 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-32B-Instruct", - "name": "Qwen2.5-Coder-32B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7265, - "hfopenllm_v2/BBH": 0.6625, - "hfopenllm_v2/MATH Level 5": 0.4955, - "hfopenllm_v2/GPQA": 0.349, - "hfopenllm_v2/MUSR": 0.4386, - "hfopenllm_v2/MMLU-PRO": 0.4413 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-7B", - "name": "Qwen2.5-Coder-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3446, - "hfopenllm_v2/BBH": 0.4856, - "hfopenllm_v2/MATH Level 5": 0.1918, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3449, - "hfopenllm_v2/MMLU-PRO": 0.3679 - } - }, - { - "id": "Qwen/Qwen2.5-Coder-7B-Instruct", - "name": "Qwen2.5-Coder-7B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6101, - "hfopenllm_v2/BBH": 0.5008, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3352 - } - }, - { - "id": "Qwen/Qwen2.5-Math-1.5B-Instruct", - "name": "Qwen2.5-Math-1.5B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.3752, - "hfopenllm_v2/MATH Level 5": 0.2628, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3685, - "hfopenllm_v2/MMLU-PRO": 0.1801 - } - }, - { - "id": "Qwen/Qwen2.5-Math-72B-Instruct", - "name": "Qwen2.5-Math-72B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4003, - "hfopenllm_v2/BBH": 0.6452, - "hfopenllm_v2/MATH Level 5": 0.6239, - "hfopenllm_v2/GPQA": 0.3314, - "hfopenllm_v2/MUSR": 0.4473, - "hfopenllm_v2/MMLU-PRO": 0.4812 - } - }, - { - "id": "Qwen/Qwen2.5-Math-7B", - "name": "Qwen2.5-Math-7B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.246, - "hfopenllm_v2/BBH": 0.4455, - "hfopenllm_v2/MATH Level 5": 0.3051, - "hfopenllm_v2/GPQA": 0.2936, - "hfopenllm_v2/MUSR": 0.3781, - "hfopenllm_v2/MMLU-PRO": 0.2718 - } - }, - { - "id": "Qwen/Qwen2.5-Math-7B-Instruct", - "name": "Qwen2.5-Math-7B-Instruct", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2636, - "hfopenllm_v2/BBH": 0.4388, - "hfopenllm_v2/MATH Level 5": 0.5808, - "hfopenllm_v2/GPQA": 0.2617, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.282 - } - }, - { - "id": "Qwen/WorldPM-72B", - "name": "Qwen/WorldPM-72B", - "developer": "Qwen", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.6333, - "reward-bench/Factuality": 0.7074, - "reward-bench/Precise IF": 0.3125, - "reward-bench/Math": 0.6557, - "reward-bench/Safety": 0.8533, - "reward-bench/Focus": 0.9172, - "reward-bench/Ties": 0.3535 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/Z.ai.json b/data/developers/Z.ai.json deleted file mode 100644 index 57c7f480b675e9b58ece7b3513ffc5f50a1b139e..0000000000000000000000000000000000000000 --- a/data/developers/Z.ai.json +++ /dev/null @@ -1,14 +0,0 @@ -{ - "developer": "Z.ai", - "models": [ - { - "id": "zhipu-ai/glm-4.6", - "name": "GLM 4.6", - "developer": "Z.ai", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 24.5 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/AALF.json b/data/developers/aalf.json similarity index 100% rename from data/developers/AALF.json rename to data/developers/aalf.json diff --git a/data/developers/Aashraf995.json b/data/developers/aashraf995.json similarity index 100% rename from data/developers/Aashraf995.json rename to data/developers/aashraf995.json diff --git a/data/developers/AbacusResearch.json b/data/developers/abacusresearch.json similarity index 100% rename from data/developers/AbacusResearch.json rename to data/developers/abacusresearch.json diff --git a/data/developers/abhishek.json b/data/developers/abhishek.json index 4a990668962b400e2663f06544d4986dc1420d7b..93669d3c565d1e56b0d776cadd09cc45571db057 100644 --- a/data/developers/abhishek.json +++ b/data/developers/abhishek.json @@ -7,12 +7,12 @@ "developer": "abhishek", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1952, - "hfopenllm_v2/BBH": 0.3127, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3584, - "hfopenllm_v2/MMLU-PRO": 0.1144 + "hfopenllm_v2/IFEval": 0.1957, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.365, + "hfopenllm_v2/MMLU-PRO": 0.1151 } }, { diff --git a/data/developers/AELLM.json b/data/developers/aellm.json similarity index 100% rename from data/developers/AELLM.json rename to data/developers/aellm.json diff --git a/data/developers/AGI-0.json b/data/developers/agi-0.json similarity index 100% rename from data/developers/AGI-0.json rename to data/developers/agi-0.json diff --git a/data/developers/Ahdoot.json b/data/developers/ahdoot.json similarity index 100% rename from data/developers/Ahdoot.json rename to data/developers/ahdoot.json diff --git a/data/developers/Ahjeong.json b/data/developers/ahjeong.json similarity index 100% rename from data/developers/Ahjeong.json rename to data/developers/ahjeong.json diff --git a/data/developers/AI-MO.json b/data/developers/ai-mo.json similarity index 100% rename from data/developers/AI-MO.json rename to data/developers/ai-mo.json diff --git a/data/developers/AI-Sweden-Models.json b/data/developers/ai-sweden-models.json similarity index 100% rename from data/developers/AI-Sweden-Models.json rename to data/developers/ai-sweden-models.json diff --git a/data/developers/ai2.json b/data/developers/ai2.json index 7e5f541744bdd2f63b0bbf1d278aa5db277042bc..8cef799e2bbc945e54e8c7aa160c4c03b21bcf7a 100644 --- a/data/developers/ai2.json +++ b/data/developers/ai2.json @@ -1,5 +1,5 @@ { - "developer": "ai2", + "developer": "AI2", "models": [ { "id": "ai2/llama-2-chat-7b-nectar-3.8m.json", @@ -43,10 +43,10 @@ "developer": "ai2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6895, + "reward-bench/Score": 0.7008, "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3706, - "reward-bench/Safety": 0.7595 + "reward-bench/Chat Hard": 0.3882, + "reward-bench/Safety": 0.7757 } }, { diff --git a/data/developers/AI4free.json b/data/developers/ai4free.json similarity index 100% rename from data/developers/AI4free.json rename to data/developers/ai4free.json diff --git a/data/developers/AicoresSecurity.json b/data/developers/aicoressecurity.json similarity index 100% rename from data/developers/AicoresSecurity.json rename to data/developers/aicoressecurity.json diff --git a/data/developers/AIDC-AI.json b/data/developers/aidc-ai.json similarity index 100% rename from data/developers/AIDC-AI.json rename to data/developers/aidc-ai.json diff --git a/data/developers/Alepach.json b/data/developers/alepach.json similarity index 100% rename from data/developers/Alepach.json rename to data/developers/alepach.json diff --git a/data/developers/AlephAlpha.json b/data/developers/alephalpha.json similarity index 100% rename from data/developers/AlephAlpha.json rename to data/developers/alephalpha.json diff --git a/data/developers/Alibaba-NLP.json b/data/developers/alibaba-nlp.json similarity index 100% rename from data/developers/Alibaba-NLP.json rename to data/developers/alibaba-nlp.json diff --git a/data/developers/alibaba.json b/data/developers/alibaba.json index 3464fe4959367b9df63ad6a26271766d68bbb5a6..844401e341e9e7b7e568fb2e5cdb8d2481e4039c 100644 --- a/data/developers/alibaba.json +++ b/data/developers/alibaba.json @@ -1,6 +1,15 @@ { - "developer": "alibaba", + "developer": "Alibaba", "models": [ + { + "id": "alibaba/qwen-3-coder-480b", + "name": "Qwen 3 Coder 480B", + "developer": "Alibaba", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 27.2 + } + }, { "id": "alibaba/qwen3-235b-a22b-instruct-2507", "name": "qwen3-235b-a22b-instruct-2507", @@ -27,6 +36,50 @@ "global-mmlu-lite/Chinese": 0.8775, "global-mmlu-lite/Burmese": 0.88 } + }, + { + "id": "alibaba/qwen3-235b-a22b-thinking-2507", + "name": "qwen3-235b-a22b-thinking-2507", + "developer": "Alibaba", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.1267605633802817, + "livecodebenchpro/Easy Problems": 0.7605633802816901 + } + }, + { + "id": "alibaba/qwen3-30b-a3b", + "name": "qwen3-30b-a3b", + "developer": "Alibaba", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.028169014084507043, + "livecodebenchpro/Easy Problems": 0.5774647887323944 + } + }, + { + "id": "alibaba/qwen3-max", + "name": "alibaba/qwen3-max", + "developer": "Alibaba", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.04225352112676056, + "livecodebenchpro/Easy Problems": 0.36619718309859156 + } + }, + { + "id": "alibaba/qwen3-next-80b-a3b-thinking", + "name": "qwen3-next-80b-a3b-thinking", + "developer": "Alibaba", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.14084507042253522, + "livecodebenchpro/Easy Problems": 0.7464788732394366 + } } ] } \ No newline at end of file diff --git a/data/developers/allenai.json b/data/developers/allenai.json index d86875077c7b6298c92ac8f5bdb15a0edeaef46e..2216de990715399db779de0d8aec537b82843775 100644 --- a/data/developers/allenai.json +++ b/data/developers/allenai.json @@ -7,17 +7,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9021, + "reward-bench/Score": 0.7606, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Safety": 0.8844, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8835 } }, { @@ -26,17 +26,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.649, - "reward-bench/Chat": 0.933, - "reward-bench/Chat Hard": 0.7785, - "reward-bench/Safety": 0.8267, - "reward-bench/Reasoning": 0.7886, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8463, "reward-bench/Factuality": 0.72, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.8851, "reward-bench/Focus": 0.8323, - "reward-bench/Ties": 0.5406 + "reward-bench/Ties": 0.5406, + "reward-bench/Chat": 0.933, + "reward-bench/Chat Hard": 0.7785, + "reward-bench/Reasoning": 0.7886, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -45,17 +45,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8885, + "reward-bench/Score": 0.7285, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8158, + "reward-bench/Safety": 0.8956, + "reward-bench/Reasoning": 0.887, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7432, "reward-bench/Precise IF": 0.4437, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.8932, "reward-bench/Focus": 0.9071, - "reward-bench/Ties": 0.7638, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8158, - "reward-bench/Reasoning": 0.887, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.7638 } }, { @@ -64,12 +64,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8291, - "hfopenllm_v2/BBH": 0.6164, - "hfopenllm_v2/MATH Level 5": 0.4502, + "hfopenllm_v2/IFEval": 0.8379, + "hfopenllm_v2/BBH": 0.6157, + "hfopenllm_v2/MATH Level 5": 0.3829, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4948, - "hfopenllm_v2/MMLU-PRO": 0.4645 + "hfopenllm_v2/MUSR": 0.4988, + "hfopenllm_v2/MMLU-PRO": 0.4656 } }, { @@ -106,17 +106,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8892, + "reward-bench/Score": 0.722, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Safety": 0.8689, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, - "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8308 } }, { @@ -125,12 +125,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { diff --git a/data/developers/Alsebay.json b/data/developers/alsebay.json similarity index 100% rename from data/developers/Alsebay.json rename to data/developers/alsebay.json diff --git a/data/developers/Amaorynho.json b/data/developers/amaorynho.json similarity index 100% rename from data/developers/Amaorynho.json rename to data/developers/amaorynho.json diff --git a/data/developers/Amu.json b/data/developers/amu.json similarity index 100% rename from data/developers/Amu.json rename to data/developers/amu.json diff --git a/data/developers/anthropic.json b/data/developers/anthropic.json index 6ddacfee09f6754246b1911e3bf4c320d5aa4eeb..988da2ba8f3164f39decf979a298c48e2f5b88c2 100644 --- a/data/developers/anthropic.json +++ b/data/developers/anthropic.json @@ -1,6 +1,60 @@ { - "developer": "anthropic", + "developer": "Anthropic", "models": [ + { + "id": "Anthropic/claude-3-5-sonnet-20240620", + "name": "Anthropic/claude-3-5-sonnet-20240620", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.8417, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.7401, + "reward-bench/Safety": 0.8162, + "reward-bench/Reasoning": 0.8469 + } + }, + { + "id": "Anthropic/claude-3-haiku-20240307", + "name": "Anthropic/claude-3-haiku-20240307", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.7289, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Safety": 0.7953, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635 + } + }, + { + "id": "Anthropic/claude-3-opus-20240229", + "name": "Anthropic/claude-3-opus-20240229", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.8008, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.8662, + "reward-bench/Reasoning": 0.7868 + } + }, + { + "id": "Anthropic/claude-3-sonnet-20240229", + "name": "Anthropic/claude-3-sonnet-20240229", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.7458, + "reward-bench/Chat": 0.9344, + "reward-bench/Chat Hard": 0.5658, + "reward-bench/Safety": 0.8169, + "reward-bench/Reasoning": 0.6907, + "reward-bench/Prior Sets (0.5 weight)": 0.6963 + } + }, { "id": "anthropic/Opus 4.1", "name": "Opus 4.1", @@ -17,8 +71,6 @@ "developer": "anthropic", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.478, - "ace/Gaming Score": 0.391, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.34, "apex-agents/Overall Mean Score": 0.348, @@ -26,6 +78,8 @@ "apex-agents/Management Consulting Pass@1": 0.132, "apex-agents/Corporate Law Pass@1": 0.202, "apex-agents/Corporate Lawyer Mean Score": 0.471, + "ace/Overall Score": 0.478, + "ace/Gaming Score": 0.391, "apex-v1/Medicine (MD) Score": 0.65 } }, @@ -540,6 +594,26 @@ "helm_mmlu/Mean win rate": 0.082 } }, + { + "id": "anthropic/claude-3.7-sonnet", + "name": "anthropic/claude-3.7-sonnet", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.014084507042253521, + "livecodebenchpro/Easy Problems": 0.15492957746478872 + } + }, + { + "id": "anthropic/claude-haiku-4.5", + "name": "Claude Haiku 4.5", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 29.8 + } + }, { "id": "anthropic/claude-instant-1.2", "name": "Claude Instant 1.2", @@ -656,6 +730,47 @@ "helm_capabilities/Omni-MATH": 0.616 } }, + { + "id": "anthropic/claude-opus-4-5", + "name": "claude-opus-4-5", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "appworld_test_normal/appworld/test_normal": 0.61, + "browsecompplus/browsecompplus": 0.61, + "swe-bench/swe-bench": 0.6061, + "tau-bench-2_airline/tau-bench-2/airline": 0.74, + "tau-bench-2_retail/tau-bench-2/retail": 0.78, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 + } + }, + { + "id": "anthropic/claude-opus-4.1", + "name": "Claude Opus 4.1", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 36.9 + } + }, + { + "id": "anthropic/claude-opus-4.5", + "name": "Claude Opus 4.5", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 52.1 + } + }, + { + "id": "anthropic/claude-opus-4.6", + "name": "Claude Opus 4.6", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 62.9 + } + }, { "id": "anthropic/claude-sonnet-4-20250514", "name": "claude-sonnet-4-20250514", @@ -721,6 +836,15 @@ "livecodebenchpro/Easy Problems": 0.5352 } }, + { + "id": "anthropic/claude-sonnet-4.5", + "name": "Claude Sonnet 4.5", + "developer": "Anthropic", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 46.5 + } + }, { "id": "anthropic/claude-v1.3", "name": "Anthropic Claude v1.3", diff --git a/data/developers/ArliAI.json b/data/developers/arliai.json similarity index 100% rename from data/developers/ArliAI.json rename to data/developers/arliai.json diff --git a/data/developers/Arthur-LAGACHERIE.json b/data/developers/arthur-lagacherie.json similarity index 100% rename from data/developers/Arthur-LAGACHERIE.json rename to data/developers/arthur-lagacherie.json diff --git a/data/developers/Artples.json b/data/developers/artples.json similarity index 100% rename from data/developers/Artples.json rename to data/developers/artples.json diff --git a/data/developers/Aryanne.json b/data/developers/aryanne.json similarity index 100% rename from data/developers/Aryanne.json rename to data/developers/aryanne.json diff --git a/data/developers/AtAndDev.json b/data/developers/atanddev.json similarity index 53% rename from data/developers/AtAndDev.json rename to data/developers/atanddev.json index d269c1fbba37e9ea4df30635a880656764997806..42530fda0188ea30162bbff58f1010442b0394c9 100644 --- a/data/developers/AtAndDev.json +++ b/data/developers/atanddev.json @@ -7,12 +7,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4511, - "hfopenllm_v2/BBH": 0.4275, - "hfopenllm_v2/MATH Level 5": 0.1473, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3623, - "hfopenllm_v2/MMLU-PRO": 0.2806 + "hfopenllm_v2/IFEval": 0.4605, + "hfopenllm_v2/BBH": 0.4258, + "hfopenllm_v2/MATH Level 5": 0.0748, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3636, + "hfopenllm_v2/MMLU-PRO": 0.2812 } } ] diff --git a/data/developers/Ateron.json b/data/developers/ateron.json similarity index 100% rename from data/developers/Ateron.json rename to data/developers/ateron.json diff --git a/data/developers/AtlaAI.json b/data/developers/atlaai.json similarity index 100% rename from data/developers/AtlaAI.json rename to data/developers/atlaai.json diff --git a/data/developers/AuraIndustries.json b/data/developers/auraindustries.json similarity index 100% rename from data/developers/AuraIndustries.json rename to data/developers/auraindustries.json diff --git a/data/developers/Aurel9.json b/data/developers/aurel9.json similarity index 100% rename from data/developers/Aurel9.json rename to data/developers/aurel9.json diff --git a/data/developers/Ayush-Singh.json b/data/developers/ayush-singh.json similarity index 100% rename from data/developers/Ayush-Singh.json rename to data/developers/ayush-singh.json diff --git a/data/developers/Azure99.json b/data/developers/azure99.json similarity index 100% rename from data/developers/Azure99.json rename to data/developers/azure99.json diff --git a/data/developers/Ba2han.json b/data/developers/ba2han.json similarity index 100% rename from data/developers/Ba2han.json rename to data/developers/ba2han.json diff --git a/data/developers/BAAI.json b/data/developers/baai.json similarity index 100% rename from data/developers/BAAI.json rename to data/developers/baai.json diff --git a/data/developers/Baptiste-HUVELLE-10.json b/data/developers/baptiste-huvelle-10.json similarity index 100% rename from data/developers/Baptiste-HUVELLE-10.json rename to data/developers/baptiste-huvelle-10.json diff --git a/data/developers/BEE-spoke-data.json b/data/developers/bee-spoke-data.json similarity index 100% rename from data/developers/BEE-spoke-data.json rename to data/developers/bee-spoke-data.json diff --git a/data/developers/BenevolenceMessiah.json b/data/developers/benevolencemessiah.json similarity index 100% rename from data/developers/BenevolenceMessiah.json rename to data/developers/benevolencemessiah.json diff --git a/data/developers/BlackBeenie.json b/data/developers/blackbeenie.json similarity index 100% rename from data/developers/BlackBeenie.json rename to data/developers/blackbeenie.json diff --git a/data/developers/Bllossom.json b/data/developers/bllossom.json similarity index 100% rename from data/developers/Bllossom.json rename to data/developers/bllossom.json diff --git a/data/developers/BoltMonkey.json b/data/developers/boltmonkey.json similarity index 100% rename from data/developers/BoltMonkey.json rename to data/developers/boltmonkey.json diff --git a/data/developers/BrainWave-ML.json b/data/developers/brainwave-ml.json similarity index 100% rename from data/developers/BrainWave-ML.json rename to data/developers/brainwave-ml.json diff --git a/data/developers/BramVanroy.json b/data/developers/bramvanroy.json similarity index 100% rename from data/developers/BramVanroy.json rename to data/developers/bramvanroy.json diff --git a/data/developers/BSC-LT.json b/data/developers/bsc-lt.json similarity index 100% rename from data/developers/BSC-LT.json rename to data/developers/bsc-lt.json diff --git a/data/developers/bunnycore.json b/data/developers/bunnycore.json index 99fe58df58a117598126b277ff0e780c1f1338e7..bb8c74d48bf152cd7e7c9c9948f93a1e957ca73a 100644 --- a/data/developers/bunnycore.json +++ b/data/developers/bunnycore.json @@ -287,12 +287,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4652, - "hfopenllm_v2/BBH": 0.4531, - "hfopenllm_v2/MATH Level 5": 0.1284, - "hfopenllm_v2/GPQA": 0.2643, - "hfopenllm_v2/MUSR": 0.3394, - "hfopenllm_v2/MMLU-PRO": 0.3152 + "hfopenllm_v2/IFEval": 0.1775, + "hfopenllm_v2/BBH": 0.295, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1049 } }, { diff --git a/data/developers/ByteDance.json b/data/developers/bytedance.json similarity index 100% rename from data/developers/ByteDance.json rename to data/developers/bytedance.json diff --git a/data/developers/CarrotAI.json b/data/developers/carrotai.json similarity index 100% rename from data/developers/CarrotAI.json rename to data/developers/carrotai.json diff --git a/data/developers/Casual-Autopsy.json b/data/developers/casual-autopsy.json similarity index 100% rename from data/developers/Casual-Autopsy.json rename to data/developers/casual-autopsy.json diff --git a/data/developers/CausalLM.json b/data/developers/causallm.json similarity index 100% rename from data/developers/CausalLM.json rename to data/developers/causallm.json diff --git a/data/developers/Changgil.json b/data/developers/changgil.json similarity index 100% rename from data/developers/Changgil.json rename to data/developers/changgil.json diff --git a/data/developers/CIR-AMS.json b/data/developers/cir-ams.json similarity index 75% rename from data/developers/CIR-AMS.json rename to data/developers/cir-ams.json index df9dcecb6f8fae5d901c2496b58813341797427e..09d0cc390a2584b96dbfe9a5acec25171fe175fa 100644 --- a/data/developers/CIR-AMS.json +++ b/data/developers/cir-ams.json @@ -7,17 +7,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } } ] diff --git a/data/developers/ClaudioItaly.json b/data/developers/claudioitaly.json similarity index 100% rename from data/developers/ClaudioItaly.json rename to data/developers/claudioitaly.json diff --git a/data/developers/cognitivecomputations.json b/data/developers/cognitivecomputations.json index 82caa40acc22796b971b12d47ba02bdfcb85f10c..7481800d75f676c41877963bea87d8b718f3e064 100644 --- a/data/developers/cognitivecomputations.json +++ b/data/developers/cognitivecomputations.json @@ -133,12 +133,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4124, - "hfopenllm_v2/BBH": 0.6383, - "hfopenllm_v2/MATH Level 5": 0.182, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4349, - "hfopenllm_v2/MMLU-PRO": 0.4525 + "hfopenllm_v2/IFEval": 0.3613, + "hfopenllm_v2/BBH": 0.6123, + "hfopenllm_v2/MATH Level 5": 0.1239, + "hfopenllm_v2/GPQA": 0.328, + "hfopenllm_v2/MUSR": 0.4112, + "hfopenllm_v2/MMLU-PRO": 0.4494 } }, { diff --git a/data/developers/CohereForAI.json b/data/developers/cohereforai.json similarity index 100% rename from data/developers/CohereForAI.json rename to data/developers/cohereforai.json diff --git a/data/developers/Columbia-NLP.json b/data/developers/columbia-nlp.json similarity index 100% rename from data/developers/Columbia-NLP.json rename to data/developers/columbia-nlp.json diff --git a/data/developers/CombinHorizon.json b/data/developers/combinhorizon.json similarity index 100% rename from data/developers/CombinHorizon.json rename to data/developers/combinhorizon.json diff --git a/data/developers/ContactDoctor.json b/data/developers/contactdoctor.json similarity index 100% rename from data/developers/ContactDoctor.json rename to data/developers/contactdoctor.json diff --git a/data/developers/ContextualAI.json b/data/developers/contextualai.json similarity index 100% rename from data/developers/ContextualAI.json rename to data/developers/contextualai.json diff --git a/data/developers/CoolSpring.json b/data/developers/coolspring.json similarity index 100% rename from data/developers/CoolSpring.json rename to data/developers/coolspring.json diff --git a/data/developers/Corianas.json b/data/developers/corianas.json similarity index 100% rename from data/developers/Corianas.json rename to data/developers/corianas.json diff --git a/data/developers/CortexLM.json b/data/developers/cortexlm.json similarity index 100% rename from data/developers/CortexLM.json rename to data/developers/cortexlm.json diff --git a/data/developers/cpayne1303.json b/data/developers/cpayne1303.json index 6d735bd94a67b9fc86d407e5a74d4ec119a21a01..878ab50b2c306395a2d661382c84d0660b0f0d51 100644 --- a/data/developers/cpayne1303.json +++ b/data/developers/cpayne1303.json @@ -35,12 +35,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1916, - "hfopenllm_v2/BBH": 0.2977, - "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/IFEval": 0.1949, + "hfopenllm_v2/BBH": 0.2965, + "hfopenllm_v2/MATH Level 5": 0.0045, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3872, - "hfopenllm_v2/MMLU-PRO": 0.1132 + "hfopenllm_v2/MUSR": 0.3885, + "hfopenllm_v2/MMLU-PRO": 0.1111 } }, { diff --git a/data/developers/Cran-May.json b/data/developers/cran-may.json similarity index 100% rename from data/developers/Cran-May.json rename to data/developers/cran-may.json diff --git a/data/developers/CreitinGameplays.json b/data/developers/creitingameplays.json similarity index 100% rename from data/developers/CreitinGameplays.json rename to data/developers/creitingameplays.json diff --git a/data/developers/CultriX.json b/data/developers/cultrix.json similarity index 100% rename from data/developers/CultriX.json rename to data/developers/cultrix.json diff --git a/data/developers/CYFRAGOVPL.json b/data/developers/cyfragovpl.json similarity index 100% rename from data/developers/CYFRAGOVPL.json rename to data/developers/cyfragovpl.json diff --git a/data/developers/Daemontatox.json b/data/developers/daemontatox.json similarity index 96% rename from data/developers/Daemontatox.json rename to data/developers/daemontatox.json index 3c61d79ec519abf4d73b74e813ac31f059bd869c..c5a7c6036734ba8e40fef4db4dfa418dcacfdf04 100644 --- a/data/developers/Daemontatox.json +++ b/data/developers/daemontatox.json @@ -35,12 +35,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4383, - "hfopenllm_v2/BBH": 0.5034, - "hfopenllm_v2/MATH Level 5": 0.1443, + "hfopenllm_v2/IFEval": 0.4398, + "hfopenllm_v2/BBH": 0.5066, + "hfopenllm_v2/MATH Level 5": 0.1488, "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4052, - "hfopenllm_v2/MMLU-PRO": 0.3778 + "hfopenllm_v2/MUSR": 0.4079, + "hfopenllm_v2/MMLU-PRO": 0.3804 } }, { @@ -231,12 +231,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3745, - "hfopenllm_v2/BBH": 0.6668, - "hfopenllm_v2/MATH Level 5": 0.4758, - "hfopenllm_v2/GPQA": 0.3943, - "hfopenllm_v2/MUSR": 0.4858, - "hfopenllm_v2/MMLU-PRO": 0.5593 + "hfopenllm_v2/IFEval": 0.4855, + "hfopenllm_v2/BBH": 0.6627, + "hfopenllm_v2/MATH Level 5": 0.4841, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4256, + "hfopenllm_v2/MMLU-PRO": 0.5542 } }, { diff --git a/data/developers/Dampfinchen.json b/data/developers/dampfinchen.json similarity index 100% rename from data/developers/Dampfinchen.json rename to data/developers/dampfinchen.json diff --git a/data/developers/Danielbrdz.json b/data/developers/danielbrdz.json similarity index 100% rename from data/developers/Danielbrdz.json rename to data/developers/danielbrdz.json diff --git a/data/developers/Dans-DiscountModels.json b/data/developers/dans-discountmodels.json similarity index 100% rename from data/developers/Dans-DiscountModels.json rename to data/developers/dans-discountmodels.json diff --git a/data/developers/Darkknight535.json b/data/developers/darkknight535.json similarity index 100% rename from data/developers/Darkknight535.json rename to data/developers/darkknight535.json diff --git a/data/developers/Databricks-Mosaic-Research.json b/data/developers/databricks-mosaic-research.json similarity index 100% rename from data/developers/Databricks-Mosaic-Research.json rename to data/developers/databricks-mosaic-research.json diff --git a/data/developers/DavidAU.json b/data/developers/davidau.json similarity index 100% rename from data/developers/DavidAU.json rename to data/developers/davidau.json diff --git a/data/developers/Davidsv.json b/data/developers/davidsv.json similarity index 100% rename from data/developers/Davidsv.json rename to data/developers/davidsv.json diff --git a/data/developers/DavieLion.json b/data/developers/davielion.json similarity index 91% rename from data/developers/DavieLion.json rename to data/developers/davielion.json index ffdc7de10295de8981ccb0c2da137caa37979e5b..2cb943ef18d08cde486079a0ba2a2bca7b4a6971 100644 --- a/data/developers/DavieLion.json +++ b/data/developers/davielion.json @@ -7,12 +7,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { diff --git a/data/developers/DebateLabKIT.json b/data/developers/debatelabkit.json similarity index 100% rename from data/developers/DebateLabKIT.json rename to data/developers/debatelabkit.json diff --git a/data/developers/Deci.json b/data/developers/deci.json similarity index 100% rename from data/developers/Deci.json rename to data/developers/deci.json diff --git a/data/developers/DeepAutoAI.json b/data/developers/deepautoai.json similarity index 100% rename from data/developers/DeepAutoAI.json rename to data/developers/deepautoai.json diff --git a/data/developers/DeepMount00.json b/data/developers/deepmount00.json similarity index 100% rename from data/developers/DeepMount00.json rename to data/developers/deepmount00.json diff --git a/data/developers/deepseek.json b/data/developers/deepseek.json index 1c17d618529a54c25c88a53026be3b63a072fc2c..ed8678bb724841c5630e0874426c48f0340ef3ea 100644 --- a/data/developers/deepseek.json +++ b/data/developers/deepseek.json @@ -1,6 +1,17 @@ { - "developer": "deepseek", + "developer": "DeepSeek", "models": [ + { + "id": "deepseek/chat-v3-0324", + "name": "deepseek/chat-v3-0324", + "developer": "DeepSeek", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.0, + "livecodebenchpro/Easy Problems": 0.19718309859154928 + } + }, { "id": "deepseek/deepseek-r1-0528", "name": "deepseek-r1-0528", @@ -54,6 +65,48 @@ "global-mmlu-lite/Chinese": 0.8161, "global-mmlu-lite/Burmese": 0.7925 } + }, + { + "id": "deepseek/deepseek-v3.2", + "name": "DeepSeek-V3.2", + "developer": "DeepSeek", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 39.6 + } + }, + { + "id": "deepseek/ep-20250214004308-p7n89", + "name": "ep-20250214004308-p7n89", + "developer": "DeepSeek", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.014084507042253521, + "livecodebenchpro/Easy Problems": 0.4225352112676056 + } + }, + { + "id": "deepseek/ep-20250228232227-z44x5", + "name": "ep-20250228232227-z44x5", + "developer": "DeepSeek", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.0, + "livecodebenchpro/Easy Problems": 0.1267605633802817 + } + }, + { + "id": "deepseek/ep-20250603132404-cgpjm", + "name": "ep-20250603132404-cgpjm", + "developer": "DeepSeek", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.08450704225352113, + "livecodebenchpro/Easy Problems": 0.5774647887323944 + } } ] } \ No newline at end of file diff --git a/data/developers/Delta-Vector.json b/data/developers/delta-vector.json similarity index 100% rename from data/developers/Delta-Vector.json rename to data/developers/delta-vector.json diff --git a/data/developers/DevQuasar.json b/data/developers/devquasar.json similarity index 100% rename from data/developers/DevQuasar.json rename to data/developers/devquasar.json diff --git a/data/developers/Dongwei.json b/data/developers/dongwei.json similarity index 100% rename from data/developers/Dongwei.json rename to data/developers/dongwei.json diff --git a/data/developers/DoppelReflEx.json b/data/developers/doppelreflex.json similarity index 100% rename from data/developers/DoppelReflEx.json rename to data/developers/doppelreflex.json diff --git a/data/developers/DreadPoor.json b/data/developers/dreadpoor.json similarity index 100% rename from data/developers/DreadPoor.json rename to data/developers/dreadpoor.json diff --git a/data/developers/DRXD1000.json b/data/developers/drxd1000.json similarity index 100% rename from data/developers/DRXD1000.json rename to data/developers/drxd1000.json diff --git a/data/developers/DUAL-GPO.json b/data/developers/dual-gpo.json similarity index 100% rename from data/developers/DUAL-GPO.json rename to data/developers/dual-gpo.json diff --git a/data/developers/DZgas.json b/data/developers/dzgas.json similarity index 100% rename from data/developers/DZgas.json rename to data/developers/dzgas.json diff --git a/data/developers/ECE-ILAB-PRYMMAL.json b/data/developers/ece-ilab-prymmal.json similarity index 100% rename from data/developers/ECE-ILAB-PRYMMAL.json rename to data/developers/ece-ilab-prymmal.json diff --git a/data/developers/Edgerunners.json b/data/developers/edgerunners.json similarity index 100% rename from data/developers/Edgerunners.json rename to data/developers/edgerunners.json diff --git a/data/developers/eleutherai.json b/data/developers/eleutherai.json index 05408b0d99003f1925545d7ec738bb633a752eea..bc54b0a53ba891d59ce8075b240ff88f2bdd86a9 100644 --- a/data/developers/eleutherai.json +++ b/data/developers/eleutherai.json @@ -1,6 +1,174 @@ { - "developer": "eleutherai", + "developer": "EleutherAI", "models": [ + { + "id": "EleutherAI/gpt-j-6b", + "name": "gpt-j-6b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2522, + "hfopenllm_v2/BBH": 0.3191, + "hfopenllm_v2/MATH Level 5": 0.0136, + "hfopenllm_v2/GPQA": 0.2458, + "hfopenllm_v2/MUSR": 0.3658, + "hfopenllm_v2/MMLU-PRO": 0.1241 + } + }, + { + "id": "EleutherAI/gpt-neo-1.3B", + "name": "gpt-neo-1.3B", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2079, + "hfopenllm_v2/BBH": 0.3039, + "hfopenllm_v2/MATH Level 5": 0.0106, + "hfopenllm_v2/GPQA": 0.2559, + "hfopenllm_v2/MUSR": 0.3817, + "hfopenllm_v2/MMLU-PRO": 0.1164 + } + }, + { + "id": "EleutherAI/gpt-neo-125m", + "name": "gpt-neo-125m", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1905, + "hfopenllm_v2/BBH": 0.3115, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3593, + "hfopenllm_v2/MMLU-PRO": 0.1026 + } + }, + { + "id": "EleutherAI/gpt-neo-2.7B", + "name": "gpt-neo-2.7B", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.259, + "hfopenllm_v2/BBH": 0.314, + "hfopenllm_v2/MATH Level 5": 0.0106, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3554, + "hfopenllm_v2/MMLU-PRO": 0.1163 + } + }, + { + "id": "EleutherAI/gpt-neox-20b", + "name": "gpt-neox-20b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2587, + "hfopenllm_v2/BBH": 0.3165, + "hfopenllm_v2/MATH Level 5": 0.0136, + "hfopenllm_v2/GPQA": 0.2433, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1155 + } + }, + { + "id": "EleutherAI/pythia-1.4b", + "name": "pythia-1.4b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2371, + "hfopenllm_v2/BBH": 0.315, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.2617, + "hfopenllm_v2/MUSR": 0.3538, + "hfopenllm_v2/MMLU-PRO": 0.1123 + } + }, + { + "id": "EleutherAI/pythia-12b", + "name": "pythia-12b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2471, + "hfopenllm_v2/BBH": 0.318, + "hfopenllm_v2/MATH Level 5": 0.0166, + "hfopenllm_v2/GPQA": 0.2466, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1109 + } + }, + { + "id": "EleutherAI/pythia-160m", + "name": "pythia-160m", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1816, + "hfopenllm_v2/BBH": 0.297, + "hfopenllm_v2/MATH Level 5": 0.0091, + "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/MUSR": 0.4179, + "hfopenllm_v2/MMLU-PRO": 0.112 + } + }, + { + "id": "EleutherAI/pythia-1b", + "name": "pythia-1b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2208, + "hfopenllm_v2/BBH": 0.3004, + "hfopenllm_v2/MATH Level 5": 0.0091, + "hfopenllm_v2/GPQA": 0.2567, + "hfopenllm_v2/MUSR": 0.3552, + "hfopenllm_v2/MMLU-PRO": 0.1136 + } + }, + { + "id": "EleutherAI/pythia-2.8b", + "name": "pythia-2.8b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2173, + "hfopenllm_v2/BBH": 0.3224, + "hfopenllm_v2/MATH Level 5": 0.0136, + "hfopenllm_v2/GPQA": 0.25, + "hfopenllm_v2/MUSR": 0.3486, + "hfopenllm_v2/MMLU-PRO": 0.1137 + } + }, + { + "id": "EleutherAI/pythia-410m", + "name": "pythia-410m", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2195, + "hfopenllm_v2/BBH": 0.3028, + "hfopenllm_v2/MATH Level 5": 0.0098, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3578, + "hfopenllm_v2/MMLU-PRO": 0.1128 + } + }, + { + "id": "EleutherAI/pythia-6.9b", + "name": "pythia-6.9b", + "developer": "EleutherAI", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2281, + "hfopenllm_v2/BBH": 0.3232, + "hfopenllm_v2/MATH Level 5": 0.0144, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3591, + "hfopenllm_v2/MMLU-PRO": 0.1147 + } + }, { "id": "eleutherai/Pythia-12B", "name": "Pythia 12B", diff --git a/data/developers/Enno-Ai.json b/data/developers/enno-ai.json similarity index 100% rename from data/developers/Enno-Ai.json rename to data/developers/enno-ai.json diff --git a/data/developers/EnnoAi.json b/data/developers/ennoai.json similarity index 100% rename from data/developers/EnnoAi.json rename to data/developers/ennoai.json diff --git a/data/developers/Epiculous.json b/data/developers/epiculous.json similarity index 100% rename from data/developers/Epiculous.json rename to data/developers/epiculous.json diff --git a/data/developers/EpistemeAI.json b/data/developers/epistemeai.json similarity index 100% rename from data/developers/EpistemeAI.json rename to data/developers/epistemeai.json diff --git a/data/developers/EpistemeAI2.json b/data/developers/epistemeai2.json similarity index 100% rename from data/developers/EpistemeAI2.json rename to data/developers/epistemeai2.json diff --git a/data/developers/Eric111.json b/data/developers/eric111.json similarity index 100% rename from data/developers/Eric111.json rename to data/developers/eric111.json diff --git a/data/developers/Etherll.json b/data/developers/etherll.json similarity index 100% rename from data/developers/Etherll.json rename to data/developers/etherll.json diff --git a/data/developers/Eurdem.json b/data/developers/eurdem.json similarity index 100% rename from data/developers/Eurdem.json rename to data/developers/eurdem.json diff --git a/data/developers/EVA-UNIT-01.json b/data/developers/eva-unit-01.json similarity index 100% rename from data/developers/EVA-UNIT-01.json rename to data/developers/eva-unit-01.json diff --git a/data/developers/FallenMerick.json b/data/developers/fallenmerick.json similarity index 100% rename from data/developers/FallenMerick.json rename to data/developers/fallenmerick.json diff --git a/data/developers/fblgit.json b/data/developers/fblgit.json index 6c66e5b38a834fb39aa5d0ee4cfd84d4996c80a2..f27585883e30c50a326f53c796f4cb3368e9c164 100644 --- a/data/developers/fblgit.json +++ b/data/developers/fblgit.json @@ -7,12 +7,12 @@ "developer": "fblgit", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4503, - "hfopenllm_v2/BBH": 0.7035, - "hfopenllm_v2/MATH Level 5": 0.3943, - "hfopenllm_v2/GPQA": 0.401, - "hfopenllm_v2/MUSR": 0.5021, - "hfopenllm_v2/MMLU-PRO": 0.5911 + "hfopenllm_v2/IFEval": 0.5181, + "hfopenllm_v2/BBH": 0.7033, + "hfopenllm_v2/MATH Level 5": 0.4947, + "hfopenllm_v2/GPQA": 0.3826, + "hfopenllm_v2/MUSR": 0.5008, + "hfopenllm_v2/MMLU-PRO": 0.5915 } }, { diff --git a/data/developers/Felladrin.json b/data/developers/felladrin.json similarity index 100% rename from data/developers/Felladrin.json rename to data/developers/felladrin.json diff --git a/data/developers/FINGU-AI.json b/data/developers/fingu-ai.json similarity index 100% rename from data/developers/FINGU-AI.json rename to data/developers/fingu-ai.json diff --git a/data/developers/FlofloB.json b/data/developers/floflob.json similarity index 100% rename from data/developers/FlofloB.json rename to data/developers/floflob.json diff --git a/data/developers/FuJhen.json b/data/developers/fujhen.json similarity index 100% rename from data/developers/FuJhen.json rename to data/developers/fujhen.json diff --git a/data/developers/FuseAI.json b/data/developers/fuseai.json similarity index 100% rename from data/developers/FuseAI.json rename to data/developers/fuseai.json diff --git a/data/developers/GalrionSoftworks.json b/data/developers/galrionsoftworks.json similarity index 100% rename from data/developers/GalrionSoftworks.json rename to data/developers/galrionsoftworks.json diff --git a/data/developers/GenVRadmin.json b/data/developers/genvradmin.json similarity index 100% rename from data/developers/GenVRadmin.json rename to data/developers/genvradmin.json diff --git a/data/developers/Goekdeniz-Guelmez.json b/data/developers/goekdeniz-guelmez.json similarity index 90% rename from data/developers/Goekdeniz-Guelmez.json rename to data/developers/goekdeniz-guelmez.json index 55f817ee11d95051df287aacc2d790d8c57602b9..a66145ec329d3a10465fc9645b5f43caebe72067 100644 --- a/data/developers/Goekdeniz-Guelmez.json +++ b/data/developers/goekdeniz-guelmez.json @@ -7,12 +7,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3472, - "hfopenllm_v2/BBH": 0.3268, - "hfopenllm_v2/MATH Level 5": 0.0891, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3262, - "hfopenllm_v2/MMLU-PRO": 0.1641 + "hfopenllm_v2/IFEval": 0.3417, + "hfopenllm_v2/BBH": 0.3292, + "hfopenllm_v2/MATH Level 5": 0.0023, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3249, + "hfopenllm_v2/MMLU-PRO": 0.1638 } }, { @@ -133,12 +133,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7598, - "hfopenllm_v2/BBH": 0.5107, - "hfopenllm_v2/MATH Level 5": 0.4237, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.4539, - "hfopenllm_v2/MMLU-PRO": 0.4012 + "hfopenllm_v2/IFEval": 0.7628, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2802, + "hfopenllm_v2/MUSR": 0.4579, + "hfopenllm_v2/MMLU-PRO": 0.4033 } } ] diff --git a/data/developers/google.json b/data/developers/google.json index 516a0117d41131ccc1cc975646218628591986ec..4baba2f6eebc7866aaeb4423b53208bb449dbd5c 100644 --- a/data/developers/google.json +++ b/data/developers/google.json @@ -1,5 +1,5 @@ { - "developer": "google", + "developer": "Google", "models": [ { "id": "google/Gemini 2.5 Flash", @@ -28,6 +28,7 @@ "developer": "google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -35,7 +36,6 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, - "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -214,12 +214,12 @@ "developer": "google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2207, - "hfopenllm_v2/BBH": 0.4537, - "hfopenllm_v2/MATH Level 5": 0.0008, - "hfopenllm_v2/GPQA": 0.2458, - "hfopenllm_v2/MUSR": 0.422, - "hfopenllm_v2/MMLU-PRO": 0.2142 + "hfopenllm_v2/IFEval": 0.2237, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.4181, + "hfopenllm_v2/MMLU-PRO": 0.2147 } }, { @@ -924,6 +924,66 @@ "reward-bench/Ties": 0.6973 } }, + { + "id": "google/gemini-3-flash", + "name": "Gemini 3 Flash", + "developer": "Google", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 47.4 + } + }, + { + "id": "google/gemini-3-pro", + "name": "Gemini 3 Pro", + "developer": "Google", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 56.9 + } + }, + { + "id": "google/gemini-3-pro-preview", + "name": "gemini-3-pro-preview", + "developer": "Google", + "evaluator_relationship": null, + "benchmark_scores": { + "appworld_test_normal/appworld/test_normal": 0.36, + "browsecompplus/browsecompplus": 0.48, + "global-mmlu-lite/Global MMLU Lite": 0.9453, + "global-mmlu-lite/Culturally Sensitive": 0.9397, + "global-mmlu-lite/Culturally Agnostic": 0.9509, + "global-mmlu-lite/Arabic": 0.9475, + "global-mmlu-lite/English": 0.9425, + "global-mmlu-lite/Bengali": 0.9425, + "global-mmlu-lite/German": 0.94, + "global-mmlu-lite/French": 0.9575, + "global-mmlu-lite/Hindi": 0.9425, + "global-mmlu-lite/Indonesian": 0.955, + "global-mmlu-lite/Italian": 0.955, + "global-mmlu-lite/Japanese": 0.94, + "global-mmlu-lite/Korean": 0.94, + "global-mmlu-lite/Portuguese": 0.9425, + "global-mmlu-lite/Spanish": 0.9475, + "global-mmlu-lite/Swahili": 0.94, + "global-mmlu-lite/Yoruba": 0.9425, + "global-mmlu-lite/Chinese": 0.9475, + "global-mmlu-lite/Burmese": 0.9425, + "swe-bench/swe-bench": 0.71, + "tau-bench-2_airline/tau-bench-2/airline": 0.7, + "tau-bench-2_retail/tau-bench-2/retail": 0.7805, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 + } + }, + { + "id": "google/gemini-3.1-pro", + "name": "Gemini 3.1 Pro", + "developer": "Google", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 74.8 + } + }, { "id": "google/gemma-1.1-2b-it", "name": "gemma-1.1-2b-it", @@ -1144,7 +1204,8 @@ "hfopenllm_v2/MATH Level 5": 0.1949, "hfopenllm_v2/GPQA": 0.3607, "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3875 + "hfopenllm_v2/MMLU-PRO": 0.3875, + "la_leaderboard/la_leaderboard": 33.62 } }, { diff --git a/data/developers/GoToCompany.json b/data/developers/gotocompany.json similarity index 100% rename from data/developers/GoToCompany.json rename to data/developers/gotocompany.json diff --git a/data/developers/GreenNode.json b/data/developers/greennode.json similarity index 100% rename from data/developers/GreenNode.json rename to data/developers/greennode.json diff --git a/data/developers/GritLM.json b/data/developers/gritlm.json similarity index 100% rename from data/developers/GritLM.json rename to data/developers/gritlm.json diff --git a/data/developers/Groq.json b/data/developers/groq.json similarity index 100% rename from data/developers/Groq.json rename to data/developers/groq.json diff --git a/data/developers/Gryphe.json b/data/developers/gryphe.json similarity index 100% rename from data/developers/Gryphe.json rename to data/developers/gryphe.json diff --git a/data/developers/GuilhermeNaturaUmana.json b/data/developers/guilhermenaturaumana.json similarity index 100% rename from data/developers/GuilhermeNaturaUmana.json rename to data/developers/guilhermenaturaumana.json diff --git a/data/developers/Gunulhona.json b/data/developers/gunulhona.json similarity index 79% rename from data/developers/Gunulhona.json rename to data/developers/gunulhona.json index 3d63c85d80df83c6632e31f815d8dc78510d19a0..1eba4dc6aed35ea93d90c3c8c2e1a2676805ffa0 100644 --- a/data/developers/Gunulhona.json +++ b/data/developers/gunulhona.json @@ -21,12 +21,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } } ] diff --git a/data/developers/HarbingerX.json b/data/developers/harbingerx.json similarity index 100% rename from data/developers/HarbingerX.json rename to data/developers/harbingerx.json diff --git a/data/developers/Hastagaras.json b/data/developers/hastagaras.json similarity index 100% rename from data/developers/Hastagaras.json rename to data/developers/hastagaras.json diff --git a/data/developers/HelpingAI.json b/data/developers/helpingai.json similarity index 100% rename from data/developers/HelpingAI.json rename to data/developers/helpingai.json diff --git a/data/developers/hendrydong.json b/data/developers/hendrydong.json index 56e05071d3b06a1289763aa84d052b49b755b3aa..66d1cb6cae0c438dba7144b9a92c7891857c219e 100644 --- a/data/developers/hendrydong.json +++ b/data/developers/hendrydong.json @@ -7,17 +7,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7847, + "reward-bench/Score": 0.5851, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Safety": 0.6956, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, - "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508 + "reward-bench/Ties": 0.5988 } } ] diff --git a/data/developers/HeraiHench.json b/data/developers/heraihench.json similarity index 100% rename from data/developers/HeraiHench.json rename to data/developers/heraihench.json diff --git a/data/developers/HFXM.json b/data/developers/hfxm.json similarity index 100% rename from data/developers/HFXM.json rename to data/developers/hfxm.json diff --git a/data/developers/HiroseKoichi.json b/data/developers/hirosekoichi.json similarity index 100% rename from data/developers/HiroseKoichi.json rename to data/developers/hirosekoichi.json diff --git a/data/developers/HoangHa.json b/data/developers/hoangha.json similarity index 100% rename from data/developers/HoangHa.json rename to data/developers/hoangha.json diff --git a/data/developers/HPAI-BSC.json b/data/developers/hpai-bsc.json similarity index 100% rename from data/developers/HPAI-BSC.json rename to data/developers/hpai-bsc.json diff --git a/data/developers/HuggingFaceH4.json b/data/developers/huggingfaceh4.json similarity index 100% rename from data/developers/HuggingFaceH4.json rename to data/developers/huggingfaceh4.json diff --git a/data/developers/HuggingFaceTB.json b/data/developers/huggingfacetb.json similarity index 91% rename from data/developers/HuggingFaceTB.json rename to data/developers/huggingfacetb.json index bed31781473fb30427be579aab45ef01bf5054ce..912c5d4a2004a890736a0d6051d6fabe190615d4 100644 --- a/data/developers/HuggingFaceTB.json +++ b/data/developers/huggingfacetb.json @@ -133,12 +133,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0593, - "hfopenllm_v2/BBH": 0.3135, - "hfopenllm_v2/MATH Level 5": 0.0144, - "hfopenllm_v2/GPQA": 0.2341, - "hfopenllm_v2/MUSR": 0.3871, - "hfopenllm_v2/MMLU-PRO": 0.1092 + "hfopenllm_v2/IFEval": 0.2883, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.003, + "hfopenllm_v2/GPQA": 0.2357, + "hfopenllm_v2/MUSR": 0.3662, + "hfopenllm_v2/MMLU-PRO": 0.1115 } }, { @@ -161,12 +161,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } } ] diff --git a/data/developers/HumanLLMs.json b/data/developers/humanllms.json similarity index 100% rename from data/developers/HumanLLMs.json rename to data/developers/humanllms.json diff --git a/data/developers/IDEA-CCNL.json b/data/developers/idea-ccnl.json similarity index 100% rename from data/developers/IDEA-CCNL.json rename to data/developers/idea-ccnl.json diff --git a/data/developers/iFaz.json b/data/developers/ifaz.json similarity index 100% rename from data/developers/iFaz.json rename to data/developers/ifaz.json diff --git a/data/developers/IlyaGusev.json b/data/developers/ilyagusev.json similarity index 100% rename from data/developers/IlyaGusev.json rename to data/developers/ilyagusev.json diff --git a/data/developers/Infinirc.json b/data/developers/infinirc.json similarity index 100% rename from data/developers/Infinirc.json rename to data/developers/infinirc.json diff --git a/data/developers/infly.json b/data/developers/infly.json index d497bf1e2632542284e99f21127cba81c8ed1b97..fe3f0dc6f7a4b2c08dd2895544fd05de4f16df3c 100644 --- a/data/developers/infly.json +++ b/data/developers/infly.json @@ -7,16 +7,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9511, + "reward-bench/Score": 0.7648, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Safety": 0.9644, + "reward-bench/Reasoning": 0.9912, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Reasoning": 0.9912 + "reward-bench/Ties": 0.8622 } } ] diff --git a/data/developers/INSAIT-Institute.json b/data/developers/insait-institute.json similarity index 100% rename from data/developers/INSAIT-Institute.json rename to data/developers/insait-institute.json diff --git a/data/developers/Intel.json b/data/developers/intel.json similarity index 100% rename from data/developers/Intel.json rename to data/developers/intel.json diff --git a/data/developers/internlm.json b/data/developers/internlm.json index 69708dbd584389b0c656d19a5340943c92e82210..1712d52ae0cee8c4072f0e9a63a1b0ae3dc99827 100644 --- a/data/developers/internlm.json +++ b/data/developers/internlm.json @@ -21,16 +21,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -39,16 +39,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5628, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Safety": 0.6111, - "reward-bench/Reasoning": 0.9576, + "reward-bench/Score": 0.9016, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, + "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483 + "reward-bench/Ties": 0.5483, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Reasoning": 0.9576 } }, { diff --git a/data/developers/IntervitensInc.json b/data/developers/intervitensinc.json similarity index 100% rename from data/developers/IntervitensInc.json rename to data/developers/intervitensinc.json diff --git a/data/developers/Invalid-Null.json b/data/developers/invalid-null.json similarity index 100% rename from data/developers/Invalid-Null.json rename to data/developers/invalid-null.json diff --git a/data/developers/iRyanBell.json b/data/developers/iryanbell.json similarity index 100% rename from data/developers/iRyanBell.json rename to data/developers/iryanbell.json diff --git a/data/developers/Isaak-Carter.json b/data/developers/isaak-carter.json similarity index 84% rename from data/developers/Isaak-Carter.json rename to data/developers/isaak-carter.json index efbc1cacd0b6bbfbc287a808cc9dcc2a866d50e8..b50bb29eeb6bc3b6ea2e8e9ad1635c8f8be99880 100644 --- a/data/developers/Isaak-Carter.json +++ b/data/developers/isaak-carter.json @@ -7,12 +7,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2553, - "hfopenllm_v2/BBH": 0.4725, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.3654, - "hfopenllm_v2/MMLU-PRO": 0.3316 + "hfopenllm_v2/IFEval": 0.2477, + "hfopenllm_v2/BBH": 0.4758, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.3641, + "hfopenllm_v2/MMLU-PRO": 0.3292 } }, { diff --git a/data/developers/J-LAB.json b/data/developers/j-lab.json similarity index 100% rename from data/developers/J-LAB.json rename to data/developers/j-lab.json diff --git a/data/developers/JackFram.json b/data/developers/jackfram.json similarity index 100% rename from data/developers/JackFram.json rename to data/developers/jackfram.json diff --git a/data/developers/Jacoby746.json b/data/developers/jacoby746.json similarity index 100% rename from data/developers/Jacoby746.json rename to data/developers/jacoby746.json diff --git a/data/developers/jaspionjader.json b/data/developers/jaspionjader.json index a117f106f2a5d6899d3f145088878651c642288f..baf443efdf6d8e1ac63b7fec859014ab20d3aa12 100644 --- a/data/developers/jaspionjader.json +++ b/data/developers/jaspionjader.json @@ -119,12 +119,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4345, - "hfopenllm_v2/BBH": 0.5419, - "hfopenllm_v2/MATH Level 5": 0.1292, - "hfopenllm_v2/GPQA": 0.3087, + "hfopenllm_v2/IFEval": 0.4418, + "hfopenllm_v2/BBH": 0.5406, + "hfopenllm_v2/MATH Level 5": 0.1352, + "hfopenllm_v2/GPQA": 0.3062, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.3854 + "hfopenllm_v2/MMLU-PRO": 0.386 } }, { diff --git a/data/developers/JayHyeon.json b/data/developers/jayhyeon.json similarity index 100% rename from data/developers/JayHyeon.json rename to data/developers/jayhyeon.json diff --git a/data/developers/Jimmy19991222.json b/data/developers/jimmy19991222.json similarity index 100% rename from data/developers/Jimmy19991222.json rename to data/developers/jimmy19991222.json diff --git a/data/developers/Joseph717171.json b/data/developers/joseph717171.json similarity index 100% rename from data/developers/Joseph717171.json rename to data/developers/joseph717171.json diff --git a/data/developers/Josephgflowers.json b/data/developers/josephgflowers.json similarity index 100% rename from data/developers/Josephgflowers.json rename to data/developers/josephgflowers.json diff --git a/data/developers/JungZoona.json b/data/developers/jungzoona.json similarity index 100% rename from data/developers/JungZoona.json rename to data/developers/jungzoona.json diff --git a/data/developers/Junhoee.json b/data/developers/junhoee.json similarity index 100% rename from data/developers/Junhoee.json rename to data/developers/junhoee.json diff --git a/data/developers/Khetterman.json b/data/developers/khetterman.json similarity index 100% rename from data/developers/Khetterman.json rename to data/developers/khetterman.json diff --git a/data/developers/Kimargin.json b/data/developers/kimargin.json similarity index 100% rename from data/developers/Kimargin.json rename to data/developers/kimargin.json diff --git a/data/developers/Kimi.json b/data/developers/kimi.json similarity index 100% rename from data/developers/Kimi.json rename to data/developers/kimi.json diff --git a/data/developers/KingNish.json b/data/developers/kingnish.json similarity index 100% rename from data/developers/KingNish.json rename to data/developers/kingnish.json diff --git a/data/developers/Kquant03.json b/data/developers/kquant03.json similarity index 100% rename from data/developers/Kquant03.json rename to data/developers/kquant03.json diff --git a/data/developers/Krystalan.json b/data/developers/krystalan.json similarity index 100% rename from data/developers/Krystalan.json rename to data/developers/krystalan.json diff --git a/data/developers/KSU-HW-SEC.json b/data/developers/ksu-hw-sec.json similarity index 100% rename from data/developers/KSU-HW-SEC.json rename to data/developers/ksu-hw-sec.json diff --git a/data/developers/Kuaishou.json b/data/developers/kuaishou.json similarity index 100% rename from data/developers/Kuaishou.json rename to data/developers/kuaishou.json diff --git a/data/developers/Kukedlc.json b/data/developers/kukedlc.json similarity index 100% rename from data/developers/Kukedlc.json rename to data/developers/kukedlc.json diff --git a/data/developers/Kumar955.json b/data/developers/kumar955.json similarity index 100% rename from data/developers/Kumar955.json rename to data/developers/kumar955.json diff --git a/data/developers/L-RAGE.json b/data/developers/l-rage.json similarity index 100% rename from data/developers/L-RAGE.json rename to data/developers/l-rage.json diff --git a/data/developers/Lambent.json b/data/developers/lambent.json similarity index 100% rename from data/developers/Lambent.json rename to data/developers/lambent.json diff --git a/data/developers/Langboat.json b/data/developers/langboat.json similarity index 100% rename from data/developers/Langboat.json rename to data/developers/langboat.json diff --git a/data/developers/Lawnakk.json b/data/developers/lawnakk.json similarity index 100% rename from data/developers/Lawnakk.json rename to data/developers/lawnakk.json diff --git a/data/developers/LEESM.json b/data/developers/leesm.json similarity index 100% rename from data/developers/LEESM.json rename to data/developers/leesm.json diff --git a/data/developers/LenguajeNaturalAI.json b/data/developers/lenguajenaturalai.json similarity index 100% rename from data/developers/LenguajeNaturalAI.json rename to data/developers/lenguajenaturalai.json diff --git a/data/developers/LeroyDyer.json b/data/developers/leroydyer.json similarity index 100% rename from data/developers/LeroyDyer.json rename to data/developers/leroydyer.json diff --git a/data/developers/LGAI-EXAONE.json b/data/developers/lgai-exaone.json similarity index 100% rename from data/developers/LGAI-EXAONE.json rename to data/developers/lgai-exaone.json diff --git a/data/developers/LightningRodLabs.json b/data/developers/lightningrodlabs.json similarity index 100% rename from data/developers/LightningRodLabs.json rename to data/developers/lightningrodlabs.json diff --git a/data/developers/Lil-R.json b/data/developers/lil-r.json similarity index 100% rename from data/developers/Lil-R.json rename to data/developers/lil-r.json diff --git a/data/developers/LilRg.json b/data/developers/lilrg.json similarity index 100% rename from data/developers/LilRg.json rename to data/developers/lilrg.json diff --git a/data/developers/LimYeri.json b/data/developers/limyeri.json similarity index 100% rename from data/developers/LimYeri.json rename to data/developers/limyeri.json diff --git a/data/developers/LLM360.json b/data/developers/llm360.json similarity index 100% rename from data/developers/LLM360.json rename to data/developers/llm360.json diff --git a/data/developers/LLM4Binary.json b/data/developers/llm4binary.json similarity index 100% rename from data/developers/LLM4Binary.json rename to data/developers/llm4binary.json diff --git a/data/developers/llnYou.json b/data/developers/llnyou.json similarity index 100% rename from data/developers/llnYou.json rename to data/developers/llnyou.json diff --git a/data/developers/Locutusque.json b/data/developers/locutusque.json similarity index 100% rename from data/developers/Locutusque.json rename to data/developers/locutusque.json diff --git a/data/developers/Luni.json b/data/developers/luni.json similarity index 100% rename from data/developers/Luni.json rename to data/developers/luni.json diff --git a/data/developers/Lunzima.json b/data/developers/lunzima.json similarity index 100% rename from data/developers/Lunzima.json rename to data/developers/lunzima.json diff --git a/data/developers/LxzGordon.json b/data/developers/lxzgordon.json similarity index 85% rename from data/developers/LxzGordon.json rename to data/developers/lxzgordon.json index e4ace3cc8f7193c8c711403c980535ee73fdd6d3..7f802cf733857054e01537f3ecf745a3fdb38a05 100644 --- a/data/developers/LxzGordon.json +++ b/data/developers/lxzgordon.json @@ -20,16 +20,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9294, + "reward-bench/Score": 0.7394, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Safety": 0.9178, + "reward-bench/Reasoning": 0.9698, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Reasoning": 0.9698 + "reward-bench/Ties": 0.7653 } } ] diff --git a/data/developers/Lyte.json b/data/developers/lyte.json similarity index 100% rename from data/developers/Lyte.json rename to data/developers/lyte.json diff --git a/data/developers/M4-ai.json b/data/developers/m4-ai.json similarity index 100% rename from data/developers/M4-ai.json rename to data/developers/m4-ai.json diff --git a/data/developers/Magpie-Align.json b/data/developers/magpie-align.json similarity index 93% rename from data/developers/Magpie-Align.json rename to data/developers/magpie-align.json index 155a416e715db19186e6af681f6736ea9ed101d1..0dc0a43e89bb456a61006caa30051add55effb08 100644 --- a/data/developers/Magpie-Align.json +++ b/data/developers/magpie-align.json @@ -35,12 +35,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4118, - "hfopenllm_v2/BBH": 0.4811, - "hfopenllm_v2/MATH Level 5": 0.034, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3047, - "hfopenllm_v2/MMLU-PRO": 0.3006 + "hfopenllm_v2/IFEval": 0.4027, + "hfopenllm_v2/BBH": 0.4789, + "hfopenllm_v2/MATH Level 5": 0.0461, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.3087, + "hfopenllm_v2/MMLU-PRO": 0.3001 } }, { diff --git a/data/developers/MagusCorp.json b/data/developers/maguscorp.json similarity index 100% rename from data/developers/MagusCorp.json rename to data/developers/maguscorp.json diff --git a/data/developers/ManoloPueblo.json b/data/developers/manolopueblo.json similarity index 100% rename from data/developers/ManoloPueblo.json rename to data/developers/manolopueblo.json diff --git a/data/developers/MarinaraSpaghetti.json b/data/developers/marinaraspaghetti.json similarity index 100% rename from data/developers/MarinaraSpaghetti.json rename to data/developers/marinaraspaghetti.json diff --git a/data/developers/Marsouuu.json b/data/developers/marsouuu.json similarity index 100% rename from data/developers/Marsouuu.json rename to data/developers/marsouuu.json diff --git a/data/developers/matouLeLoup.json b/data/developers/matouleloup.json similarity index 100% rename from data/developers/matouLeLoup.json rename to data/developers/matouleloup.json diff --git a/data/developers/MaziyarPanahi.json b/data/developers/maziyarpanahi.json similarity index 100% rename from data/developers/MaziyarPanahi.json rename to data/developers/maziyarpanahi.json diff --git a/data/developers/meraGPT.json b/data/developers/meragpt.json similarity index 100% rename from data/developers/meraGPT.json rename to data/developers/meragpt.json diff --git a/data/developers/MEscriva.json b/data/developers/mescriva.json similarity index 100% rename from data/developers/MEscriva.json rename to data/developers/mescriva.json diff --git a/data/developers/meta.json b/data/developers/meta.json index de736e49b4d41c787178b80ac40ea100d79fc1c1..26bc17a2e391a382c21d56412db070e8aafb4bed 100644 --- a/data/developers/meta.json +++ b/data/developers/meta.json @@ -1,5 +1,5 @@ { - "developer": "meta", + "developer": "Meta", "models": [ { "id": "meta/LLaMA-13B", @@ -820,6 +820,17 @@ "helm_mmlu/Mean win rate": 0.722 } }, + { + "id": "meta/llama-4-maverick", + "name": "meta/llama-4-maverick", + "developer": "Meta", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.0, + "livecodebenchpro/Easy Problems": 0.09859154929577464 + } + }, { "id": "meta/llama-4-maverick-17b-128e-instruct-fp8", "name": "Llama 4 Maverick 17Bx128E Instruct FP8", diff --git a/data/developers/microsoft.json b/data/developers/microsoft.json index 8f4a0b4aafd0e6a24d113b83750e80b0a606d1be..fed5ebc923dff7f3b9e889e69e470163179f9050 100644 --- a/data/developers/microsoft.json +++ b/data/developers/microsoft.json @@ -417,12 +417,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0585, - "hfopenllm_v2/BBH": 0.6691, - "hfopenllm_v2/MATH Level 5": 0.3165, - "hfopenllm_v2/GPQA": 0.406, + "hfopenllm_v2/IFEval": 0.0488, + "hfopenllm_v2/BBH": 0.6703, + "hfopenllm_v2/MATH Level 5": 0.2787, + "hfopenllm_v2/GPQA": 0.401, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5287 + "hfopenllm_v2/MMLU-PRO": 0.5295 } } ] diff --git a/data/developers/migtissera.json b/data/developers/migtissera.json index 758dddd03daa89044ae5d24105c8c1c0de081425..1e70db18b5c73c29853f73048efc3f85a3c53fba 100644 --- a/data/developers/migtissera.json +++ b/data/developers/migtissera.json @@ -105,12 +105,12 @@ "developer": "migtissera", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.443, - "hfopenllm_v2/BBH": 0.5706, - "hfopenllm_v2/MATH Level 5": 0.0869, - "hfopenllm_v2/GPQA": 0.3079, - "hfopenllm_v2/MUSR": 0.4031, - "hfopenllm_v2/MMLU-PRO": 0.3354 + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5686, + "hfopenllm_v2/MATH Level 5": 0.0838, + "hfopenllm_v2/GPQA": 0.3003, + "hfopenllm_v2/MUSR": 0.4045, + "hfopenllm_v2/MMLU-PRO": 0.334 } } ] diff --git a/data/developers/Minami-su.json b/data/developers/minami-su.json similarity index 100% rename from data/developers/Minami-su.json rename to data/developers/minami-su.json diff --git a/data/developers/minimax.json b/data/developers/minimax.json index 4109f62ad72b732d20b6d28221c6abd3e74fcf6a..3eb98fb6a6609e5f1cc4d74a47ecc2b74aaa9bb0 100644 --- a/data/developers/minimax.json +++ b/data/developers/minimax.json @@ -1,5 +1,5 @@ { - "developer": "minimax", + "developer": "MiniMax", "models": [ { "id": "minimax/Minimax-2.5", @@ -9,6 +9,33 @@ "benchmark_scores": { "apex-agents/Corporate Lawyer Mean Score": 0.339 } + }, + { + "id": "minimax/minimax-m2", + "name": "MiniMax M2", + "developer": "MiniMax", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 30.0 + } + }, + { + "id": "minimax/minimax-m2.1", + "name": "MiniMax M2.1", + "developer": "MiniMax", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 36.6 + } + }, + { + "id": "minimax/minimax-m2.5", + "name": "Minimax m2.5", + "developer": "Minimax", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 42.2 + } } ] } \ No newline at end of file diff --git a/data/developers/mistralai.json b/data/developers/mistralai.json index 49ed42eb9fa248702f2e286105ac1814c51dacf3..9e08c45340505201318f8ac58a31b423a7da6a56 100644 --- a/data/developers/mistralai.json +++ b/data/developers/mistralai.json @@ -246,12 +246,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2326, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.4413, - "hfopenllm_v2/MMLU-PRO": 0.3871 + "hfopenllm_v2/IFEval": 0.2415, + "hfopenllm_v2/BBH": 0.5087, + "hfopenllm_v2/MATH Level 5": 0.102, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4321, + "hfopenllm_v2/MMLU-PRO": 0.385 } }, { diff --git a/data/developers/mlabonne.json b/data/developers/mlabonne.json index be86bd7fe732025f133b06dc7c412aa5e56b7119..2620a8c4e8931697abdcd44e4a4aae7c1e430da5 100644 --- a/data/developers/mlabonne.json +++ b/data/developers/mlabonne.json @@ -161,12 +161,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4162, - "hfopenllm_v2/BBH": 0.5124, - "hfopenllm_v2/MATH Level 5": 0.0853, - "hfopenllm_v2/GPQA": 0.3029, - "hfopenllm_v2/MUSR": 0.415, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.7561, + "hfopenllm_v2/BBH": 0.5111, + "hfopenllm_v2/MATH Level 5": 0.0906, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3841 } }, { diff --git a/data/developers/MLP-KTLim.json b/data/developers/mlp-ktlim.json similarity index 100% rename from data/developers/MLP-KTLim.json rename to data/developers/mlp-ktlim.json diff --git a/data/developers/ModelCloud.json b/data/developers/modelcloud.json similarity index 100% rename from data/developers/ModelCloud.json rename to data/developers/modelcloud.json diff --git a/data/developers/ModelSpace.json b/data/developers/modelspace.json similarity index 100% rename from data/developers/ModelSpace.json rename to data/developers/modelspace.json diff --git a/data/developers/MoonRide.json b/data/developers/moonride.json similarity index 100% rename from data/developers/MoonRide.json rename to data/developers/moonride.json diff --git a/data/developers/Moonshot_AI.json b/data/developers/moonshot_ai.json similarity index 100% rename from data/developers/Moonshot_AI.json rename to data/developers/moonshot_ai.json diff --git a/data/developers/Mostafa8Mehrabi.json b/data/developers/mostafa8mehrabi.json similarity index 100% rename from data/developers/Mostafa8Mehrabi.json rename to data/developers/mostafa8mehrabi.json diff --git a/data/developers/MrRobotoAI.json b/data/developers/mrrobotoai.json similarity index 100% rename from data/developers/MrRobotoAI.json rename to data/developers/mrrobotoai.json diff --git a/data/developers/MTSAIR.json b/data/developers/mtsair.json similarity index 100% rename from data/developers/MTSAIR.json rename to data/developers/mtsair.json diff --git a/data/developers/Multiple.json b/data/developers/multiple.json similarity index 80% rename from data/developers/Multiple.json rename to data/developers/multiple.json index b464ffc342f6679111ee9b59dfbf7cef7cf4cfc7..34cdb844d495e12fd3a3820204fbda313306e211 100644 --- a/data/developers/Multiple.json +++ b/data/developers/multiple.json @@ -7,7 +7,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 50.1 + "terminal-bench-2.0/terminal-bench-2.0": 72.4 } } ] diff --git a/data/developers/MultivexAI.json b/data/developers/multivexai.json similarity index 100% rename from data/developers/MultivexAI.json rename to data/developers/multivexai.json diff --git a/data/developers/Mxode.json b/data/developers/mxode.json similarity index 100% rename from data/developers/Mxode.json rename to data/developers/mxode.json diff --git a/data/developers/NAPS-ai.json b/data/developers/naps-ai.json similarity index 100% rename from data/developers/NAPS-ai.json rename to data/developers/naps-ai.json diff --git a/data/developers/Naveenpoliasetty.json b/data/developers/naveenpoliasetty.json similarity index 100% rename from data/developers/Naveenpoliasetty.json rename to data/developers/naveenpoliasetty.json diff --git a/data/developers/NbAiLab.json b/data/developers/nbailab.json similarity index 100% rename from data/developers/NbAiLab.json rename to data/developers/nbailab.json diff --git a/data/developers/NCSOFT.json b/data/developers/ncsoft.json similarity index 89% rename from data/developers/NCSOFT.json rename to data/developers/ncsoft.json index 5cc19c52d2bf4d85f74e73a3056f8da048c42869..78886a7e33419ae72417f31ea13b0a41a2a3ced1 100644 --- a/data/developers/NCSOFT.json +++ b/data/developers/ncsoft.json @@ -20,16 +20,16 @@ "developer": "NCSOFT", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.648, - "reward-bench/Chat": 0.9721, - "reward-bench/Chat Hard": 0.818, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9192, + "reward-bench/Score": 0.8942, "reward-bench/Factuality": 0.6084, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.9596, - "reward-bench/Ties": 0.6786 + "reward-bench/Ties": 0.6786, + "reward-bench/Chat": 0.9721, + "reward-bench/Chat Hard": 0.818, + "reward-bench/Reasoning": 0.9192 } }, { diff --git a/data/developers/Nekochu.json b/data/developers/nekochu.json similarity index 100% rename from data/developers/Nekochu.json rename to data/developers/nekochu.json diff --git a/data/developers/NeverSleep.json b/data/developers/neversleep.json similarity index 100% rename from data/developers/NeverSleep.json rename to data/developers/neversleep.json diff --git a/data/developers/Nexesenex.json b/data/developers/nexesenex.json similarity index 100% rename from data/developers/Nexesenex.json rename to data/developers/nexesenex.json diff --git a/data/developers/Nexusflow.json b/data/developers/nexusflow.json similarity index 85% rename from data/developers/Nexusflow.json rename to data/developers/nexusflow.json index 2f78cab8780e463261ce35ef251c295eb3f3fd0f..49fe739e7104f59b148ecb24e067de8a0cb500b5 100644 --- a/data/developers/Nexusflow.json +++ b/data/developers/nexusflow.json @@ -21,17 +21,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8133, + "reward-bench/Score": 0.4553, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7556, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137 + "reward-bench/Ties": 0.1004 } } ] diff --git a/data/developers/nicolinho.json b/data/developers/nicolinho.json index 551d4a5de698babd0e830b509f51bb11f4dd2ac7..79bf445ae201aa8b9add0559d92e4abd4fd3bebb 100644 --- a/data/developers/nicolinho.json +++ b/data/developers/nicolinho.json @@ -7,16 +7,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9444, + "reward-bench/Score": 0.7667, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Safety": 0.9578, + "reward-bench/Reasoning": 0.9826, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Reasoning": 0.9826 + "reward-bench/Ties": 0.8321 } }, { @@ -51,16 +51,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9314, + "reward-bench/Score": 0.7074, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Safety": 0.9467, + "reward-bench/Reasoning": 0.9677, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, - "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Reasoning": 0.9677 + "reward-bench/Ties": 0.7234 } } ] diff --git a/data/developers/NikolaSigmoid.json b/data/developers/nikolasigmoid.json similarity index 100% rename from data/developers/NikolaSigmoid.json rename to data/developers/nikolasigmoid.json diff --git a/data/developers/nisten.json b/data/developers/nisten.json index 7b275a3c64fa267662b1c7ec09c2c6db9c0fbfc6..785709badefb68361515ca958f0d589a8268f14c 100644 --- a/data/developers/nisten.json +++ b/data/developers/nisten.json @@ -7,12 +7,12 @@ "developer": "nisten", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3914, - "hfopenllm_v2/BBH": 0.6591, - "hfopenllm_v2/MATH Level 5": 0.3044, - "hfopenllm_v2/GPQA": 0.3591, - "hfopenllm_v2/MUSR": 0.4681, - "hfopenllm_v2/MMLU-PRO": 0.5611 + "hfopenllm_v2/IFEval": 0.3799, + "hfopenllm_v2/BBH": 0.6647, + "hfopenllm_v2/MATH Level 5": 0.3406, + "hfopenllm_v2/GPQA": 0.4035, + "hfopenllm_v2/MUSR": 0.494, + "hfopenllm_v2/MMLU-PRO": 0.5731 } }, { diff --git a/data/developers/Nitral-AI.json b/data/developers/nitral-ai.json similarity index 100% rename from data/developers/Nitral-AI.json rename to data/developers/nitral-ai.json diff --git a/data/developers/NJS26.json b/data/developers/njs26.json similarity index 100% rename from data/developers/NJS26.json rename to data/developers/njs26.json diff --git a/data/developers/NLPark.json b/data/developers/nlpark.json similarity index 100% rename from data/developers/NLPark.json rename to data/developers/nlpark.json diff --git a/data/developers/Nohobby.json b/data/developers/nohobby.json similarity index 100% rename from data/developers/Nohobby.json rename to data/developers/nohobby.json diff --git a/data/developers/Norquinal.json b/data/developers/norquinal.json similarity index 100% rename from data/developers/Norquinal.json rename to data/developers/norquinal.json diff --git a/data/developers/NotASI.json b/data/developers/notasi.json similarity index 100% rename from data/developers/NotASI.json rename to data/developers/notasi.json diff --git a/data/developers/NousResearch.json b/data/developers/nousresearch.json similarity index 100% rename from data/developers/NousResearch.json rename to data/developers/nousresearch.json diff --git a/data/developers/Novaciano.json b/data/developers/novaciano.json similarity index 100% rename from data/developers/Novaciano.json rename to data/developers/novaciano.json diff --git a/data/developers/NTQAI.json b/data/developers/ntqai.json similarity index 100% rename from data/developers/NTQAI.json rename to data/developers/ntqai.json diff --git a/data/developers/NucleusAI.json b/data/developers/nucleusai.json similarity index 100% rename from data/developers/NucleusAI.json rename to data/developers/nucleusai.json diff --git a/data/developers/NYTK.json b/data/developers/nytk.json similarity index 100% rename from data/developers/NYTK.json rename to data/developers/nytk.json diff --git a/data/developers/NyxKrage.json b/data/developers/nyxkrage.json similarity index 100% rename from data/developers/NyxKrage.json rename to data/developers/nyxkrage.json diff --git a/data/developers/OEvortex.json b/data/developers/oevortex.json similarity index 100% rename from data/developers/OEvortex.json rename to data/developers/oevortex.json diff --git a/data/developers/OliveiraJLT.json b/data/developers/oliveirajlt.json similarity index 100% rename from data/developers/OliveiraJLT.json rename to data/developers/oliveirajlt.json diff --git a/data/developers/Omkar1102.json b/data/developers/omkar1102.json similarity index 58% rename from data/developers/Omkar1102.json rename to data/developers/omkar1102.json index 1d044781189744d770af993cfdb651c1e02eee6f..ca0270b469e58068ee5242b6801e2fbe782ddc0a 100644 --- a/data/developers/Omkar1102.json +++ b/data/developers/omkar1102.json @@ -7,12 +7,12 @@ "developer": "Omkar1102", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2254, - "hfopenllm_v2/BBH": 0.275, + "hfopenllm_v2/IFEval": 0.2148, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3762, - "hfopenllm_v2/MMLU-PRO": 0.1123 + "hfopenllm_v2/GPQA": 0.2508, + "hfopenllm_v2/MUSR": 0.3802, + "hfopenllm_v2/MMLU-PRO": 0.1126 } } ] diff --git a/data/developers/OmnicromsBrain.json b/data/developers/omnicromsbrain.json similarity index 100% rename from data/developers/OmnicromsBrain.json rename to data/developers/omnicromsbrain.json diff --git a/data/developers/OnlyCheeini.json b/data/developers/onlycheeini.json similarity index 100% rename from data/developers/OnlyCheeini.json rename to data/developers/onlycheeini.json diff --git a/data/developers/ontocord.json b/data/developers/ontocord.json index 26bf0b3ef88f6bd021717109acdee1567c9357e3..42739e95ae91594f9d3e1b7009d00eb7697a5ad4 100644 --- a/data/developers/ontocord.json +++ b/data/developers/ontocord.json @@ -273,12 +273,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { diff --git a/data/developers/oopere.json b/data/developers/oopere.json index 71d39c6c35063e6c97b2a688c804f115030e73dd..31f8ae539b90532ab5313131a90014d14bcc6621 100644 --- a/data/developers/oopere.json +++ b/data/developers/oopere.json @@ -7,12 +7,12 @@ "developer": "oopere", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2164, - "hfopenllm_v2/BBH": 0.3169, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/IFEval": 0.2119, + "hfopenllm_v2/BBH": 0.3156, + "hfopenllm_v2/MATH Level 5": 0.0181, + "hfopenllm_v2/GPQA": 0.2567, "hfopenllm_v2/MUSR": 0.3832, - "hfopenllm_v2/MMLU-PRO": 0.1134 + "hfopenllm_v2/MMLU-PRO": 0.113 } }, { diff --git a/data/developers/Open-Orca.json b/data/developers/open-orca.json similarity index 100% rename from data/developers/Open-Orca.json rename to data/developers/open-orca.json diff --git a/data/developers/openai.json b/data/developers/openai.json index b8a74d2a5687da3273bb20273947cddb36eec00c..cf19213ab810b1dcc8b6e097736ee0c3702dbc0c 100644 --- a/data/developers/openai.json +++ b/data/developers/openai.json @@ -1,5 +1,5 @@ { - "developer": "openai", + "developer": "OpenAI", "models": [ { "id": "openai/GPT 4o", @@ -613,6 +613,17 @@ "reward-bench/Prior Sets (0.5 weight)": 0.7363 } }, + { + "id": "openai/gpt-4.1", + "name": "openai/gpt-4.1", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "livecodebenchpro/Hard Problems": 0.0, + "livecodebenchpro/Medium Problems": 0.0, + "livecodebenchpro/Easy Problems": 0.19718309859154928 + } + }, { "id": "openai/gpt-4.1-2025-04-14", "name": "gpt-4.1-2025-04-14", @@ -894,16 +905,25 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 + } + }, + { + "id": "openai/gpt-5", + "name": "GPT-5", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 35.2 } }, { @@ -942,6 +962,24 @@ "livecodebenchpro/Easy Problems": 0.8873239436619719 } }, + { + "id": "openai/gpt-5-codex", + "name": "GPT-5-Codex", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 41.3 + } + }, + { + "id": "openai/gpt-5-mini", + "name": "GPT-5-Mini", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 29.2 + } + }, { "id": "openai/gpt-5-mini-2025-08-07", "name": "GPT-5 mini 2025-08-07", @@ -956,6 +994,15 @@ "helm_capabilities/Omni-MATH": 0.722 } }, + { + "id": "openai/gpt-5-nano", + "name": "GPT-5-Nano", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 7.9 + } + }, { "id": "openai/gpt-5-nano-2025-08-07", "name": "GPT-5 nano 2025-08-07", @@ -970,6 +1017,86 @@ "helm_capabilities/Omni-MATH": 0.547 } }, + { + "id": "openai/gpt-5.1", + "name": "GPT-5.1", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 47.6 + } + }, + { + "id": "openai/gpt-5.1-codex", + "name": "GPT-5.1-Codex", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 53.5 + } + }, + { + "id": "openai/gpt-5.1-codex-max", + "name": "GPT-5.1-Codex-Max", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 60.4 + } + }, + { + "id": "openai/gpt-5.1-codex-mini", + "name": "GPT-5.1-Codex-Mini", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 43.1 + } + }, + { + "id": "openai/gpt-5.2", + "name": "GPT-5.2", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 60.7 + } + }, + { + "id": "openai/gpt-5.2-2025-12-11", + "name": "gpt-5.2-2025-12-11", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "appworld_test_normal/appworld/test_normal": 0.22, + "browsecompplus/browsecompplus": 0.46, + "livecodebenchpro/Hard Problems": 0.1594, + "livecodebenchpro/Medium Problems": 0.5211, + "livecodebenchpro/Easy Problems": 0.9014, + "swe-bench/swe-bench": 0.5253, + "tau-bench-2_airline/tau-bench-2/airline": 0.6, + "tau-bench-2_retail/tau-bench-2/retail": 0.51, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 + } + }, + { + "id": "openai/gpt-5.2-codex", + "name": "GPT-5.2-Codex", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 66.5 + } + }, + { + "id": "openai/gpt-5.3-codex", + "name": "GPT-5.3-Codex", + "developer": "OpenAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 77.3 + } + }, { "id": "openai/gpt-oss-120b", "name": "gpt-oss-120b", @@ -985,7 +1112,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.11267605633802817, "livecodebenchpro/Easy Problems": 0.6619718309859155, - "terminal-bench-2.0/terminal-bench-2.0": 18.7 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } }, { @@ -1106,9 +1233,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.0143, - "livecodebenchpro/Medium Problems": 0.2923, - "livecodebenchpro/Easy Problems": 0.8571 + "livecodebenchpro/Hard Problems": 0.014084507042253521, + "livecodebenchpro/Medium Problems": 0.30985915492957744, + "livecodebenchpro/Easy Problems": 0.8873239436619719 } }, { diff --git a/data/developers/OpenAssistant.json b/data/developers/openassistant.json similarity index 93% rename from data/developers/OpenAssistant.json rename to data/developers/openassistant.json index 01c75aa86e6780446770cc9407ec4e8d8baaeaf2..ad2c9bbf5e049296e5d368971f1365bc748e1226 100644 --- a/data/developers/OpenAssistant.json +++ b/data/developers/openassistant.json @@ -59,17 +59,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.32, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Safety": 0.3667, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836, + "reward-bench/Score": 0.6126, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, + "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12 + "reward-bench/Ties": 0.12, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836 } } ] diff --git a/data/developers/openbmb.json b/data/developers/openbmb.json index dd2ac9b2651a2fee615e466e3b3da81ac3cfeef4..d8dae84054074ce01b5c47fc58b69a148fdc99c0 100644 --- a/data/developers/openbmb.json +++ b/data/developers/openbmb.json @@ -21,17 +21,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5806, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Safety": 0.6267, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172, + "reward-bench/Score": 0.8159, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972 + "reward-bench/Ties": 0.5972, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172 } }, { @@ -68,17 +68,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6903, + "reward-bench/Score": 0.4683, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Safety": 0.5089, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, - "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294 + "reward-bench/Ties": 0.3036 } } ] diff --git a/data/developers/OpenBuddy.json b/data/developers/openbuddy.json similarity index 100% rename from data/developers/OpenBuddy.json rename to data/developers/openbuddy.json diff --git a/data/developers/OpenGenerativeAI.json b/data/developers/opengenerativeai.json similarity index 100% rename from data/developers/OpenGenerativeAI.json rename to data/developers/opengenerativeai.json diff --git a/data/developers/OpenLeecher.json b/data/developers/openleecher.json similarity index 100% rename from data/developers/OpenLeecher.json rename to data/developers/openleecher.json diff --git a/data/developers/OpenLLM-France.json b/data/developers/openllm-france.json similarity index 100% rename from data/developers/OpenLLM-France.json rename to data/developers/openllm-france.json diff --git a/data/developers/OpenScholar.json b/data/developers/openscholar.json similarity index 100% rename from data/developers/OpenScholar.json rename to data/developers/openscholar.json diff --git a/data/developers/Orenguteng.json b/data/developers/orenguteng.json similarity index 100% rename from data/developers/Orenguteng.json rename to data/developers/orenguteng.json diff --git a/data/developers/Orion-zhen.json b/data/developers/orion-zhen.json similarity index 100% rename from data/developers/Orion-zhen.json rename to data/developers/orion-zhen.json diff --git a/data/developers/P0x0.json b/data/developers/p0x0.json similarity index 100% rename from data/developers/P0x0.json rename to data/developers/p0x0.json diff --git a/data/developers/Parissa3.json b/data/developers/parissa3.json similarity index 100% rename from data/developers/Parissa3.json rename to data/developers/parissa3.json diff --git a/data/developers/Pinkstack.json b/data/developers/pinkstack.json similarity index 100% rename from data/developers/Pinkstack.json rename to data/developers/pinkstack.json diff --git a/data/developers/PJMixers-Dev.json b/data/developers/pjmixers-dev.json similarity index 100% rename from data/developers/PJMixers-Dev.json rename to data/developers/pjmixers-dev.json diff --git a/data/developers/PJMixers.json b/data/developers/pjmixers.json similarity index 100% rename from data/developers/PJMixers.json rename to data/developers/pjmixers.json diff --git a/data/developers/PKU-Alignment.json b/data/developers/pku-alignment.json similarity index 84% rename from data/developers/PKU-Alignment.json rename to data/developers/pku-alignment.json index c49cf2d2c5ee4c838ff715dbd555f36543ad5f7c..76e1f41b6171c4fd2a3d35d25df175a75a77a416 100644 --- a/data/developers/PKU-Alignment.json +++ b/data/developers/pku-alignment.json @@ -7,17 +7,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3332, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Safety": 0.7589, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57, + "reward-bench/Score": 0.5798, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, + "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57 } }, { @@ -26,17 +26,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4727, + "reward-bench/Score": 0.1606, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Safety": 0.1422, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, - "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993 + "reward-bench/Ties": -0.01 } }, { @@ -45,17 +45,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3326, - "reward-bench/Chat": 0.5726, - "reward-bench/Chat Hard": 0.4561, - "reward-bench/Safety": 0.7356, - "reward-bench/Reasoning": 0.6211, - "reward-bench/Prior Sets (0.5 weight)": 0.5397, + "reward-bench/Score": 0.5957, "reward-bench/Factuality": 0.3789, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3333, + "reward-bench/Safety": 0.7608, "reward-bench/Focus": 0.2828, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.5726, + "reward-bench/Chat Hard": 0.4561, + "reward-bench/Reasoning": 0.6211, + "reward-bench/Prior Sets (0.5 weight)": 0.5397 } }, { diff --git a/data/developers/PocketDoc.json b/data/developers/pocketdoc.json similarity index 100% rename from data/developers/PocketDoc.json rename to data/developers/pocketdoc.json diff --git a/data/developers/PoLL.json b/data/developers/poll.json similarity index 100% rename from data/developers/PoLL.json rename to data/developers/poll.json diff --git a/data/developers/PowerInfer.json b/data/developers/powerinfer.json similarity index 100% rename from data/developers/PowerInfer.json rename to data/developers/powerinfer.json diff --git a/data/developers/PranavHarshan.json b/data/developers/pranavharshan.json similarity index 100% rename from data/developers/PranavHarshan.json rename to data/developers/pranavharshan.json diff --git a/data/developers/Pretergeek.json b/data/developers/pretergeek.json similarity index 100% rename from data/developers/Pretergeek.json rename to data/developers/pretergeek.json diff --git a/data/developers/PrimeIntellect.json b/data/developers/primeintellect.json similarity index 100% rename from data/developers/PrimeIntellect.json rename to data/developers/primeintellect.json diff --git a/data/developers/prithivMLmods.json b/data/developers/prithivmlmods.json similarity index 100% rename from data/developers/prithivMLmods.json rename to data/developers/prithivmlmods.json diff --git a/data/developers/PuxAI.json b/data/developers/puxai.json similarity index 100% rename from data/developers/PuxAI.json rename to data/developers/puxai.json diff --git a/data/developers/PygmalionAI.json b/data/developers/pygmalionai.json similarity index 100% rename from data/developers/PygmalionAI.json rename to data/developers/pygmalionai.json diff --git a/data/developers/Q-bert.json b/data/developers/q-bert.json similarity index 100% rename from data/developers/Q-bert.json rename to data/developers/q-bert.json diff --git a/data/developers/qingy2019.json b/data/developers/qingy2019.json index 9e607d380bbf6837078a684f50c178b1c83bab9f..3885f54f7f780ebe87cee5b8aacc1c1136f4441f 100644 --- a/data/developers/qingy2019.json +++ b/data/developers/qingy2019.json @@ -35,12 +35,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2401, - "hfopenllm_v2/BBH": 0.4622, - "hfopenllm_v2/MATH Level 5": 0.0725, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3703, - "hfopenllm_v2/MMLU-PRO": 0.2379 + "hfopenllm_v2/IFEval": 0.2358, + "hfopenllm_v2/BBH": 0.4612, + "hfopenllm_v2/MATH Level 5": 0.0642, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3717, + "hfopenllm_v2/MMLU-PRO": 0.2382 } }, { @@ -49,12 +49,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { diff --git a/data/developers/Quazim0t0.json b/data/developers/quazim0t0.json similarity index 99% rename from data/developers/Quazim0t0.json rename to data/developers/quazim0t0.json index 2d35a4f5e966e8ec3da190137c155e9ea4458c4d..0303ee2938913b3a08d6ca1dcd55193c92e3b91f 100644 --- a/data/developers/Quazim0t0.json +++ b/data/developers/quazim0t0.json @@ -77,12 +77,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6718, - "hfopenllm_v2/BBH": 0.6891, - "hfopenllm_v2/MATH Level 5": 0.4985, - "hfopenllm_v2/GPQA": 0.3339, - "hfopenllm_v2/MUSR": 0.4323, - "hfopenllm_v2/MMLU-PRO": 0.5408 + "hfopenllm_v2/IFEval": 0.6654, + "hfopenllm_v2/BBH": 0.6901, + "hfopenllm_v2/MATH Level 5": 0.4698, + "hfopenllm_v2/GPQA": 0.3331, + "hfopenllm_v2/MUSR": 0.431, + "hfopenllm_v2/MMLU-PRO": 0.5426 } }, { diff --git a/data/developers/qwen.json b/data/developers/qwen.json index 97558c520ad4a20eeac87e7c81dc7120c0e9c64f..6a8f6299becf13918917b4fe59b8db0f09fec0ca 100644 --- a/data/developers/qwen.json +++ b/data/developers/qwen.json @@ -1,6 +1,884 @@ { - "developer": "qwen", + "developer": "Qwen", "models": [ + { + "id": "Qwen/QwQ-32B", + "name": "QwQ-32B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3977, + "hfopenllm_v2/BBH": 0.2983, + "hfopenllm_v2/MATH Level 5": 0.1609, + "hfopenllm_v2/GPQA": 0.2601, + "hfopenllm_v2/MUSR": 0.4206, + "hfopenllm_v2/MMLU-PRO": 0.1196 + } + }, + { + "id": "Qwen/QwQ-32B-Preview", + "name": "QwQ-32B-Preview", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4035, + "hfopenllm_v2/BBH": 0.6691, + "hfopenllm_v2/MATH Level 5": 0.4494, + "hfopenllm_v2/GPQA": 0.2819, + "hfopenllm_v2/MUSR": 0.411, + "hfopenllm_v2/MMLU-PRO": 0.5678 + } + }, + { + "id": "Qwen/Qwen1.5-0.5B", + "name": "Qwen1.5-0.5B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1706, + "hfopenllm_v2/BBH": 0.3154, + "hfopenllm_v2/MATH Level 5": 0.0174, + "hfopenllm_v2/GPQA": 0.2542, + "hfopenllm_v2/MUSR": 0.3616, + "hfopenllm_v2/MMLU-PRO": 0.1307 + } + }, + { + "id": "Qwen/Qwen1.5-0.5B-Chat", + "name": "Qwen1.5-0.5B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1807, + "hfopenllm_v2/BBH": 0.3167, + "hfopenllm_v2/MATH Level 5": 0.0068, + "hfopenllm_v2/GPQA": 0.2693, + "hfopenllm_v2/MUSR": 0.3837, + "hfopenllm_v2/MMLU-PRO": 0.1213, + "reward-bench/Score": 0.5298, + "reward-bench/Chat": 0.3547, + "reward-bench/Chat Hard": 0.6294, + "reward-bench/Safety": 0.5703, + "reward-bench/Reasoning": 0.5984, + "reward-bench/Prior Sets (0.5 weight)": 0.4629 + } + }, + { + "id": "Qwen/Qwen1.5-1.8B", + "name": "Qwen1.5-1.8B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2154, + "hfopenllm_v2/BBH": 0.3476, + "hfopenllm_v2/MATH Level 5": 0.0317, + "hfopenllm_v2/GPQA": 0.3054, + "hfopenllm_v2/MUSR": 0.3605, + "hfopenllm_v2/MMLU-PRO": 0.1882 + } + }, + { + "id": "Qwen/Qwen1.5-1.8B-Chat", + "name": "Qwen1.5-1.8B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2019, + "hfopenllm_v2/BBH": 0.3256, + "hfopenllm_v2/MATH Level 5": 0.0196, + "hfopenllm_v2/GPQA": 0.2978, + "hfopenllm_v2/MUSR": 0.426, + "hfopenllm_v2/MMLU-PRO": 0.1804, + "reward-bench/Score": 0.589, + "reward-bench/Chat": 0.5615, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Safety": 0.4838, + "reward-bench/Reasoning": 0.7793, + "reward-bench/Prior Sets (0.5 weight)": 0.4453 + } + }, + { + "id": "Qwen/Qwen1.5-110B", + "name": "Qwen1.5-110B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3422, + "hfopenllm_v2/BBH": 0.61, + "hfopenllm_v2/MATH Level 5": 0.247, + "hfopenllm_v2/GPQA": 0.3523, + "hfopenllm_v2/MUSR": 0.4408, + "hfopenllm_v2/MMLU-PRO": 0.5361 + } + }, + { + "id": "Qwen/Qwen1.5-110B-Chat", + "name": "Qwen1.5-110B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5939, + "hfopenllm_v2/BBH": 0.6184, + "hfopenllm_v2/MATH Level 5": 0.2341, + "hfopenllm_v2/GPQA": 0.3414, + "hfopenllm_v2/MUSR": 0.4522, + "hfopenllm_v2/MMLU-PRO": 0.4825 + } + }, + { + "id": "Qwen/Qwen1.5-14B", + "name": "Qwen1.5-14B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2905, + "hfopenllm_v2/BBH": 0.508, + "hfopenllm_v2/MATH Level 5": 0.2024, + "hfopenllm_v2/GPQA": 0.2945, + "hfopenllm_v2/MUSR": 0.4186, + "hfopenllm_v2/MMLU-PRO": 0.3644 + } + }, + { + "id": "Qwen/Qwen1.5-14B-Chat", + "name": "Qwen1.5-14B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4768, + "hfopenllm_v2/BBH": 0.5229, + "hfopenllm_v2/MATH Level 5": 0.1526, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.44, + "hfopenllm_v2/MMLU-PRO": 0.3618, + "reward-bench/Score": 0.6864, + "reward-bench/Chat": 0.5726, + "reward-bench/Chat Hard": 0.7018, + "reward-bench/Safety": 0.7122, + "reward-bench/Reasoning": 0.8961, + "reward-bench/Prior Sets (0.5 weight)": 0.4123 + } + }, + { + "id": "Qwen/Qwen1.5-32B", + "name": "Qwen1.5-32B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3297, + "hfopenllm_v2/BBH": 0.5715, + "hfopenllm_v2/MATH Level 5": 0.3029, + "hfopenllm_v2/GPQA": 0.3297, + "hfopenllm_v2/MUSR": 0.4278, + "hfopenllm_v2/MMLU-PRO": 0.45 + } + }, + { + "id": "Qwen/Qwen1.5-32B-Chat", + "name": "Qwen1.5-32B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5532, + "hfopenllm_v2/BBH": 0.6067, + "hfopenllm_v2/MATH Level 5": 0.1956, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.416, + "hfopenllm_v2/MMLU-PRO": 0.4457 + } + }, + { + "id": "Qwen/Qwen1.5-4B", + "name": "Qwen1.5-4B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2445, + "hfopenllm_v2/BBH": 0.4054, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.3604, + "hfopenllm_v2/MMLU-PRO": 0.246 + } + }, + { + "id": "Qwen/Qwen1.5-4B-Chat", + "name": "Qwen1.5-4B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3157, + "hfopenllm_v2/BBH": 0.4006, + "hfopenllm_v2/MATH Level 5": 0.0279, + "hfopenllm_v2/GPQA": 0.2668, + "hfopenllm_v2/MUSR": 0.3978, + "hfopenllm_v2/MMLU-PRO": 0.2396, + "reward-bench/Score": 0.5477, + "reward-bench/Chat": 0.3883, + "reward-bench/Chat Hard": 0.6272, + "reward-bench/Safety": 0.5568, + "reward-bench/Reasoning": 0.6689, + "reward-bench/Prior Sets (0.5 weight)": 0.447 + } + }, + { + "id": "Qwen/Qwen1.5-72B-Chat", + "name": "Qwen/Qwen1.5-72B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.6723, + "reward-bench/Chat": 0.6229, + "reward-bench/Chat Hard": 0.6601, + "reward-bench/Safety": 0.6757, + "reward-bench/Reasoning": 0.8554, + "reward-bench/Prior Sets (0.5 weight)": 0.4226 + } + }, + { + "id": "Qwen/Qwen1.5-7B", + "name": "Qwen1.5-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2684, + "hfopenllm_v2/BBH": 0.456, + "hfopenllm_v2/MATH Level 5": 0.0929, + "hfopenllm_v2/GPQA": 0.2987, + "hfopenllm_v2/MUSR": 0.4103, + "hfopenllm_v2/MMLU-PRO": 0.2916 + } + }, + { + "id": "Qwen/Qwen1.5-7B-Chat", + "name": "Qwen1.5-7B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4371, + "hfopenllm_v2/BBH": 0.451, + "hfopenllm_v2/MATH Level 5": 0.0627, + "hfopenllm_v2/GPQA": 0.3029, + "hfopenllm_v2/MUSR": 0.3779, + "hfopenllm_v2/MMLU-PRO": 0.2951, + "reward-bench/Score": 0.675, + "reward-bench/Chat": 0.5363, + "reward-bench/Chat Hard": 0.6908, + "reward-bench/Safety": 0.6919, + "reward-bench/Reasoning": 0.9041, + "reward-bench/Prior Sets (0.5 weight)": 0.4288 + } + }, + { + "id": "Qwen/Qwen1.5-MoE-A2.7B", + "name": "Qwen1.5-MoE-A2.7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.266, + "hfopenllm_v2/BBH": 0.4114, + "hfopenllm_v2/MATH Level 5": 0.0929, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.4013, + "hfopenllm_v2/MMLU-PRO": 0.2778 + } + }, + { + "id": "Qwen/Qwen1.5-MoE-A2.7B-Chat", + "name": "Qwen1.5-MoE-A2.7B-Chat", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3795, + "hfopenllm_v2/BBH": 0.4272, + "hfopenllm_v2/MATH Level 5": 0.0634, + "hfopenllm_v2/GPQA": 0.2743, + "hfopenllm_v2/MUSR": 0.3899, + "hfopenllm_v2/MMLU-PRO": 0.2923, + "reward-bench/Score": 0.6644, + "reward-bench/Chat": 0.7291, + "reward-bench/Chat Hard": 0.6316, + "reward-bench/Safety": 0.6284, + "reward-bench/Reasoning": 0.774, + "reward-bench/Prior Sets (0.5 weight)": 0.4536 + } + }, + { + "id": "Qwen/Qwen2-0.5B", + "name": "Qwen2-0.5B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1873, + "hfopenllm_v2/BBH": 0.3239, + "hfopenllm_v2/MATH Level 5": 0.0264, + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3752, + "hfopenllm_v2/MMLU-PRO": 0.172 + } + }, + { + "id": "Qwen/Qwen2-0.5B-Instruct", + "name": "Qwen2-0.5B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2247, + "hfopenllm_v2/BBH": 0.3173, + "hfopenllm_v2/MATH Level 5": 0.0287, + "hfopenllm_v2/GPQA": 0.2466, + "hfopenllm_v2/MUSR": 0.3353, + "hfopenllm_v2/MMLU-PRO": 0.1531 + } + }, + { + "id": "Qwen/Qwen2-1.5B", + "name": "Qwen2-1.5B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2113, + "hfopenllm_v2/BBH": 0.3575, + "hfopenllm_v2/MATH Level 5": 0.0702, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3658, + "hfopenllm_v2/MMLU-PRO": 0.2552 + } + }, + { + "id": "Qwen/Qwen2-1.5B-Instruct", + "name": "Qwen2-1.5B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3371, + "hfopenllm_v2/BBH": 0.3852, + "hfopenllm_v2/MATH Level 5": 0.0718, + "hfopenllm_v2/GPQA": 0.2617, + "hfopenllm_v2/MUSR": 0.4293, + "hfopenllm_v2/MMLU-PRO": 0.2501 + } + }, + { + "id": "Qwen/Qwen2-57B-A14B", + "name": "Qwen2-57B-A14B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3113, + "hfopenllm_v2/BBH": 0.5618, + "hfopenllm_v2/MATH Level 5": 0.1866, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4174, + "hfopenllm_v2/MMLU-PRO": 0.4916 + } + }, + { + "id": "Qwen/Qwen2-57B-A14B-Instruct", + "name": "Qwen2-57B-A14B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.6338, + "hfopenllm_v2/BBH": 0.5888, + "hfopenllm_v2/MATH Level 5": 0.2817, + "hfopenllm_v2/GPQA": 0.3314, + "hfopenllm_v2/MUSR": 0.4361, + "hfopenllm_v2/MMLU-PRO": 0.4575 + } + }, + { + "id": "Qwen/Qwen2-72B", + "name": "Qwen2-72B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3824, + "hfopenllm_v2/BBH": 0.6617, + "hfopenllm_v2/MATH Level 5": 0.3112, + "hfopenllm_v2/GPQA": 0.3943, + "hfopenllm_v2/MUSR": 0.4704, + "hfopenllm_v2/MMLU-PRO": 0.5731 + } + }, + { + "id": "Qwen/Qwen2-72B-Instruct", + "name": "Qwen2-72B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.7989, + "hfopenllm_v2/BBH": 0.6977, + "hfopenllm_v2/MATH Level 5": 0.4177, + "hfopenllm_v2/GPQA": 0.3725, + "hfopenllm_v2/MUSR": 0.456, + "hfopenllm_v2/MMLU-PRO": 0.5403 + } + }, + { + "id": "Qwen/Qwen2-7B", + "name": "Qwen2-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3149, + "hfopenllm_v2/BBH": 0.5315, + "hfopenllm_v2/MATH Level 5": 0.2039, + "hfopenllm_v2/GPQA": 0.3045, + "hfopenllm_v2/MUSR": 0.4439, + "hfopenllm_v2/MMLU-PRO": 0.4183 + } + }, + { + "id": "Qwen/Qwen2-7B-Instruct", + "name": "Qwen2-7B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5679, + "hfopenllm_v2/BBH": 0.5545, + "hfopenllm_v2/MATH Level 5": 0.2764, + "hfopenllm_v2/GPQA": 0.2978, + "hfopenllm_v2/MUSR": 0.3928, + "hfopenllm_v2/MMLU-PRO": 0.3847 + } + }, + { + "id": "Qwen/Qwen2-Math-72B-Instruct", + "name": "Qwen2-Math-72B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5694, + "hfopenllm_v2/BBH": 0.6343, + "hfopenllm_v2/MATH Level 5": 0.5536, + "hfopenllm_v2/GPQA": 0.3683, + "hfopenllm_v2/MUSR": 0.4517, + "hfopenllm_v2/MMLU-PRO": 0.4273 + } + }, + { + "id": "Qwen/Qwen2-Math-7B", + "name": "Qwen2-Math-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2687, + "hfopenllm_v2/BBH": 0.387, + "hfopenllm_v2/MATH Level 5": 0.2477, + "hfopenllm_v2/GPQA": 0.2634, + "hfopenllm_v2/MUSR": 0.3593, + "hfopenllm_v2/MMLU-PRO": 0.1197 + } + }, + { + "id": "Qwen/Qwen2-VL-72B-Instruct", + "name": "Qwen2-VL-72B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.5982, + "hfopenllm_v2/BBH": 0.6946, + "hfopenllm_v2/MATH Level 5": 0.3444, + "hfopenllm_v2/GPQA": 0.3876, + "hfopenllm_v2/MUSR": 0.4492, + "hfopenllm_v2/MMLU-PRO": 0.5717 + } + }, + { + "id": "Qwen/Qwen2-VL-7B-Instruct", + "name": "Qwen2-VL-7B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4599, + "hfopenllm_v2/BBH": 0.5465, + "hfopenllm_v2/MATH Level 5": 0.1986, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.4375, + "hfopenllm_v2/MMLU-PRO": 0.4095 + } + }, + { + "id": "Qwen/Qwen2.5-0.5B", + "name": "Qwen2.5-0.5B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1627, + "hfopenllm_v2/BBH": 0.3275, + "hfopenllm_v2/MATH Level 5": 0.0393, + "hfopenllm_v2/GPQA": 0.2466, + "hfopenllm_v2/MUSR": 0.3433, + "hfopenllm_v2/MMLU-PRO": 0.1906 + } + }, + { + "id": "Qwen/Qwen2.5-0.5B-Instruct", + "name": "Qwen2.5-0.5B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3153, + "hfopenllm_v2/BBH": 0.3322, + "hfopenllm_v2/MATH Level 5": 0.1035, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3342, + "hfopenllm_v2/MMLU-PRO": 0.172 + } + }, + { + "id": "Qwen/Qwen2.5-1.5B", + "name": "Qwen2.5-1.5B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2674, + "hfopenllm_v2/BBH": 0.4078, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2852, + "hfopenllm_v2/MUSR": 0.3576, + "hfopenllm_v2/MMLU-PRO": 0.2855 + } + }, + { + "id": "Qwen/Qwen2.5-1.5B-Instruct", + "name": "Qwen2.5-1.5B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4476, + "hfopenllm_v2/BBH": 0.4289, + "hfopenllm_v2/MATH Level 5": 0.2205, + "hfopenllm_v2/GPQA": 0.2559, + "hfopenllm_v2/MUSR": 0.3663, + "hfopenllm_v2/MMLU-PRO": 0.2799 + } + }, + { + "id": "Qwen/Qwen2.5-14B", + "name": "Qwen2.5-14B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3694, + "hfopenllm_v2/BBH": 0.6161, + "hfopenllm_v2/MATH Level 5": 0.29, + "hfopenllm_v2/GPQA": 0.3817, + "hfopenllm_v2/MUSR": 0.4502, + "hfopenllm_v2/MMLU-PRO": 0.5249 + } + }, + { + "id": "Qwen/Qwen2.5-14B-Instruct", + "name": "Qwen2.5-14B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.8158, + "hfopenllm_v2/BBH": 0.639, + "hfopenllm_v2/MATH Level 5": 0.5476, + "hfopenllm_v2/GPQA": 0.3221, + "hfopenllm_v2/MUSR": 0.4101, + "hfopenllm_v2/MMLU-PRO": 0.4904 + } + }, + { + "id": "Qwen/Qwen2.5-14B-Instruct-1M", + "name": "Qwen2.5-14B-Instruct-1M", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.8414, + "hfopenllm_v2/BBH": 0.6198, + "hfopenllm_v2/MATH Level 5": 0.5302, + "hfopenllm_v2/GPQA": 0.3431, + "hfopenllm_v2/MUSR": 0.418, + "hfopenllm_v2/MMLU-PRO": 0.485 + } + }, + { + "id": "Qwen/Qwen2.5-32B", + "name": "Qwen2.5-32B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4077, + "hfopenllm_v2/BBH": 0.6771, + "hfopenllm_v2/MATH Level 5": 0.3565, + "hfopenllm_v2/GPQA": 0.4119, + "hfopenllm_v2/MUSR": 0.4978, + "hfopenllm_v2/MMLU-PRO": 0.5805 + } + }, + { + "id": "Qwen/Qwen2.5-32B-Instruct", + "name": "Qwen2.5-32B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.8346, + "hfopenllm_v2/BBH": 0.6913, + "hfopenllm_v2/MATH Level 5": 0.6254, + "hfopenllm_v2/GPQA": 0.3381, + "hfopenllm_v2/MUSR": 0.4261, + "hfopenllm_v2/MMLU-PRO": 0.5667 + } + }, + { + "id": "Qwen/Qwen2.5-3B", + "name": "Qwen2.5-3B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.269, + "hfopenllm_v2/BBH": 0.4612, + "hfopenllm_v2/MATH Level 5": 0.148, + "hfopenllm_v2/GPQA": 0.2978, + "hfopenllm_v2/MUSR": 0.4303, + "hfopenllm_v2/MMLU-PRO": 0.3203 + } + }, + { + "id": "Qwen/Qwen2.5-3B-Instruct", + "name": "Qwen2.5-3B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.6475, + "hfopenllm_v2/BBH": 0.4693, + "hfopenllm_v2/MATH Level 5": 0.3678, + "hfopenllm_v2/GPQA": 0.2727, + "hfopenllm_v2/MUSR": 0.3968, + "hfopenllm_v2/MMLU-PRO": 0.3255 + } + }, + { + "id": "Qwen/Qwen2.5-72B", + "name": "Qwen2.5-72B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4137, + "hfopenllm_v2/BBH": 0.6797, + "hfopenllm_v2/MATH Level 5": 0.3912, + "hfopenllm_v2/GPQA": 0.4052, + "hfopenllm_v2/MUSR": 0.4771, + "hfopenllm_v2/MMLU-PRO": 0.5968 + } + }, + { + "id": "Qwen/Qwen2.5-72B-Instruct", + "name": "Qwen2.5-72B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.8638, + "hfopenllm_v2/BBH": 0.7273, + "hfopenllm_v2/MATH Level 5": 0.5982, + "hfopenllm_v2/GPQA": 0.375, + "hfopenllm_v2/MUSR": 0.4206, + "hfopenllm_v2/MMLU-PRO": 0.5626 + } + }, + { + "id": "Qwen/Qwen2.5-7B", + "name": "Qwen2.5-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3374, + "hfopenllm_v2/BBH": 0.5416, + "hfopenllm_v2/MATH Level 5": 0.2508, + "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/MUSR": 0.4424, + "hfopenllm_v2/MMLU-PRO": 0.4365, + "la_leaderboard/la_leaderboard": 27.61 + } + }, + { + "id": "Qwen/Qwen2.5-7B-Instruct", + "name": "Qwen2.5-7B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.7585, + "hfopenllm_v2/BBH": 0.5394, + "hfopenllm_v2/MATH Level 5": 0.5, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.402, + "hfopenllm_v2/MMLU-PRO": 0.4287 + } + }, + { + "id": "Qwen/Qwen2.5-7B-Instruct-1M", + "name": "Qwen2.5-7B-Instruct-1M", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.7448, + "hfopenllm_v2/BBH": 0.5404, + "hfopenllm_v2/MATH Level 5": 0.4335, + "hfopenllm_v2/GPQA": 0.2978, + "hfopenllm_v2/MUSR": 0.4087, + "hfopenllm_v2/MMLU-PRO": 0.3505 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-14B", + "name": "Qwen2.5-Coder-14B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3473, + "hfopenllm_v2/BBH": 0.5865, + "hfopenllm_v2/MATH Level 5": 0.2251, + "hfopenllm_v2/GPQA": 0.2928, + "hfopenllm_v2/MUSR": 0.3874, + "hfopenllm_v2/MMLU-PRO": 0.4521 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-14B-Instruct", + "name": "Qwen2.5-Coder-14B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.6908, + "hfopenllm_v2/BBH": 0.614, + "hfopenllm_v2/MATH Level 5": 0.3248, + "hfopenllm_v2/GPQA": 0.3045, + "hfopenllm_v2/MUSR": 0.3915, + "hfopenllm_v2/MMLU-PRO": 0.3939 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-32B", + "name": "Qwen2.5-Coder-32B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4363, + "hfopenllm_v2/BBH": 0.6404, + "hfopenllm_v2/MATH Level 5": 0.3089, + "hfopenllm_v2/GPQA": 0.3465, + "hfopenllm_v2/MUSR": 0.4528, + "hfopenllm_v2/MMLU-PRO": 0.5303 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-32B-Instruct", + "name": "Qwen2.5-Coder-32B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.7265, + "hfopenllm_v2/BBH": 0.6625, + "hfopenllm_v2/MATH Level 5": 0.4955, + "hfopenllm_v2/GPQA": 0.349, + "hfopenllm_v2/MUSR": 0.4386, + "hfopenllm_v2/MMLU-PRO": 0.4413 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-7B", + "name": "Qwen2.5-Coder-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.3446, + "hfopenllm_v2/BBH": 0.4856, + "hfopenllm_v2/MATH Level 5": 0.1918, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3449, + "hfopenllm_v2/MMLU-PRO": 0.3679 + } + }, + { + "id": "Qwen/Qwen2.5-Coder-7B-Instruct", + "name": "Qwen2.5-Coder-7B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.6147, + "hfopenllm_v2/BBH": 0.4999, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.3354 + } + }, + { + "id": "Qwen/Qwen2.5-Math-1.5B-Instruct", + "name": "Qwen2.5-Math-1.5B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1856, + "hfopenllm_v2/BBH": 0.3752, + "hfopenllm_v2/MATH Level 5": 0.2628, + "hfopenllm_v2/GPQA": 0.2651, + "hfopenllm_v2/MUSR": 0.3685, + "hfopenllm_v2/MMLU-PRO": 0.1801 + } + }, + { + "id": "Qwen/Qwen2.5-Math-72B-Instruct", + "name": "Qwen2.5-Math-72B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.4003, + "hfopenllm_v2/BBH": 0.6452, + "hfopenllm_v2/MATH Level 5": 0.6239, + "hfopenllm_v2/GPQA": 0.3314, + "hfopenllm_v2/MUSR": 0.4473, + "hfopenllm_v2/MMLU-PRO": 0.4812 + } + }, + { + "id": "Qwen/Qwen2.5-Math-7B", + "name": "Qwen2.5-Math-7B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.246, + "hfopenllm_v2/BBH": 0.4455, + "hfopenllm_v2/MATH Level 5": 0.3051, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.3781, + "hfopenllm_v2/MMLU-PRO": 0.2718 + } + }, + { + "id": "Qwen/Qwen2.5-Math-7B-Instruct", + "name": "Qwen2.5-Math-7B-Instruct", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.2636, + "hfopenllm_v2/BBH": 0.4388, + "hfopenllm_v2/MATH Level 5": 0.5808, + "hfopenllm_v2/GPQA": 0.2617, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.282 + } + }, + { + "id": "Qwen/WorldPM-72B", + "name": "Qwen/WorldPM-72B", + "developer": "Qwen", + "evaluator_relationship": null, + "benchmark_scores": { + "reward-bench/Score": 0.6333, + "reward-bench/Factuality": 0.7074, + "reward-bench/Precise IF": 0.3125, + "reward-bench/Math": 0.6557, + "reward-bench/Safety": 0.8533, + "reward-bench/Focus": 0.9172, + "reward-bench/Ties": 0.3535 + } + }, { "id": "qwen/qwen1.5-110b-chat", "name": "Qwen1.5 Chat 110B", diff --git a/data/developers/R-I-S-E.json b/data/developers/r-i-s-e.json similarity index 100% rename from data/developers/R-I-S-E.json rename to data/developers/r-i-s-e.json diff --git a/data/developers/Rakuten.json b/data/developers/rakuten.json similarity index 100% rename from data/developers/Rakuten.json rename to data/developers/rakuten.json diff --git a/data/developers/Ray2333.json b/data/developers/ray2333.json similarity index 97% rename from data/developers/Ray2333.json rename to data/developers/ray2333.json index f2e4e9124a06fb0ae9b827267581f8562c5b9fe1..197e293b2fa1afa6e8a7a29d93c4638037b8e1b8 100644 --- a/data/developers/Ray2333.json +++ b/data/developers/ray2333.json @@ -52,16 +52,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8839, + "reward-bench/Score": 0.5966, + "reward-bench/Chat": 0.9302, + "reward-bench/Chat Hard": 0.7719, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.912, "reward-bench/Factuality": 0.5305, "reward-bench/Precise IF": 0.3125, "reward-bench/Math": 0.5902, - "reward-bench/Safety": 0.9216, "reward-bench/Focus": 0.7455, - "reward-bench/Ties": 0.4788, - "reward-bench/Chat": 0.9302, - "reward-bench/Chat Hard": 0.7719, - "reward-bench/Reasoning": 0.912 + "reward-bench/Ties": 0.4788 } }, { diff --git a/data/developers/RDson.json b/data/developers/rdson.json similarity index 100% rename from data/developers/RDson.json rename to data/developers/rdson.json diff --git a/data/developers/recoilme.json b/data/developers/recoilme.json index 862469c1653f0d1414e3f6dabd97ef04b6d4f9e2..a767d74249450554babea0e0fb1705321ef571fe 100644 --- a/data/developers/recoilme.json +++ b/data/developers/recoilme.json @@ -7,12 +7,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2854, - "hfopenllm_v2/BBH": 0.5984, - "hfopenllm_v2/MATH Level 5": 0.1005, - "hfopenllm_v2/GPQA": 0.3297, - "hfopenllm_v2/MUSR": 0.4607, - "hfopenllm_v2/MMLU-PRO": 0.4162 + "hfopenllm_v2/IFEval": 0.7649, + "hfopenllm_v2/BBH": 0.5974, + "hfopenllm_v2/MATH Level 5": 0.0174, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4245, + "hfopenllm_v2/MMLU-PRO": 0.4207 } }, { diff --git a/data/developers/Replete-AI.json b/data/developers/replete-ai.json similarity index 95% rename from data/developers/Replete-AI.json rename to data/developers/replete-ai.json index 0f038b31b0c0cf28a02b6c25fb4fd9bd374c118c..dbb06f00736a7fcddf76b24a5c7673e098ecab57 100644 --- a/data/developers/Replete-AI.json +++ b/data/developers/replete-ai.json @@ -91,12 +91,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { diff --git a/data/developers/RESMPDEV.json b/data/developers/resmpdev.json similarity index 100% rename from data/developers/RESMPDEV.json rename to data/developers/resmpdev.json diff --git a/data/developers/RezVortex.json b/data/developers/rezvortex.json similarity index 100% rename from data/developers/RezVortex.json rename to data/developers/rezvortex.json diff --git a/data/developers/RLHFlow.json b/data/developers/rlhflow.json similarity index 100% rename from data/developers/RLHFlow.json rename to data/developers/rlhflow.json diff --git a/data/developers/Ro-xe.json b/data/developers/ro-xe.json similarity index 100% rename from data/developers/Ro-xe.json rename to data/developers/ro-xe.json diff --git a/data/developers/Rombo-Org.json b/data/developers/rombo-org.json similarity index 100% rename from data/developers/Rombo-Org.json rename to data/developers/rombo-org.json diff --git a/data/developers/rombodawg.json b/data/developers/rombodawg.json index e7b347a91d5be7f4c52661e255a50c99caa3a3fd..a241fbd73dcecc7fc768ef47719ff98c1bcb64ab 100644 --- a/data/developers/rombodawg.json +++ b/data/developers/rombodawg.json @@ -133,12 +133,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2566, - "hfopenllm_v2/BBH": 0.39, - "hfopenllm_v2/MATH Level 5": 0.1208, - "hfopenllm_v2/GPQA": 0.2626, + "hfopenllm_v2/IFEval": 0.2595, + "hfopenllm_v2/BBH": 0.3884, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2743, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2741 + "hfopenllm_v2/MMLU-PRO": 0.2719 } }, { diff --git a/data/developers/RubielLabarta.json b/data/developers/rubiellabarta.json similarity index 100% rename from data/developers/RubielLabarta.json rename to data/developers/rubiellabarta.json diff --git a/data/developers/RWKV.json b/data/developers/rwkv.json similarity index 100% rename from data/developers/RWKV.json rename to data/developers/rwkv.json diff --git a/data/developers/SaisExperiments.json b/data/developers/saisexperiments.json similarity index 100% rename from data/developers/SaisExperiments.json rename to data/developers/saisexperiments.json diff --git a/data/developers/Sakalti.json b/data/developers/sakalti.json similarity index 100% rename from data/developers/Sakalti.json rename to data/developers/sakalti.json diff --git a/data/developers/Salesforce.json b/data/developers/salesforce.json similarity index 100% rename from data/developers/Salesforce.json rename to data/developers/salesforce.json diff --git a/data/developers/SanjiWatsuki.json b/data/developers/sanjiwatsuki.json similarity index 100% rename from data/developers/SanjiWatsuki.json rename to data/developers/sanjiwatsuki.json diff --git a/data/developers/Sao10K.json b/data/developers/sao10k.json similarity index 100% rename from data/developers/Sao10K.json rename to data/developers/sao10k.json diff --git a/data/developers/Saxo.json b/data/developers/saxo.json similarity index 100% rename from data/developers/Saxo.json rename to data/developers/saxo.json diff --git a/data/developers/Schrieffer.json b/data/developers/schrieffer.json similarity index 100% rename from data/developers/Schrieffer.json rename to data/developers/schrieffer.json diff --git a/data/developers/SeaLLMs.json b/data/developers/seallms.json similarity index 100% rename from data/developers/SeaLLMs.json rename to data/developers/seallms.json diff --git a/data/developers/SenseLLM.json b/data/developers/sensellm.json similarity index 100% rename from data/developers/SenseLLM.json rename to data/developers/sensellm.json diff --git a/data/developers/SentientAGI.json b/data/developers/sentientagi.json similarity index 100% rename from data/developers/SentientAGI.json rename to data/developers/sentientagi.json diff --git a/data/developers/SeppeV.json b/data/developers/seppev.json similarity index 100% rename from data/developers/SeppeV.json rename to data/developers/seppev.json diff --git a/data/developers/SF-Foundation.json b/data/developers/sf-foundation.json similarity index 100% rename from data/developers/SF-Foundation.json rename to data/developers/sf-foundation.json diff --git a/data/developers/sfairXC.json b/data/developers/sfairxc.json similarity index 100% rename from data/developers/sfairXC.json rename to data/developers/sfairxc.json diff --git a/data/developers/Sharathhebbar24.json b/data/developers/sharathhebbar24.json similarity index 100% rename from data/developers/Sharathhebbar24.json rename to data/developers/sharathhebbar24.json diff --git a/data/developers/ShikaiChen.json b/data/developers/shikaichen.json similarity index 76% rename from data/developers/ShikaiChen.json rename to data/developers/shikaichen.json index 6162cba6dec7eaed27af88272ffb98342af2522b..6502ef5b0e02cc85862537458f25816eb0826b7d 100644 --- a/data/developers/ShikaiChen.json +++ b/data/developers/shikaichen.json @@ -7,16 +7,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } } ] diff --git a/data/developers/Shreyash2010.json b/data/developers/shreyash2010.json similarity index 100% rename from data/developers/Shreyash2010.json rename to data/developers/shreyash2010.json diff --git a/data/developers/Sicarius-Prototyping.json b/data/developers/sicarius-prototyping.json similarity index 100% rename from data/developers/Sicarius-Prototyping.json rename to data/developers/sicarius-prototyping.json diff --git a/data/developers/SicariusSicariiStuff.json b/data/developers/sicariussicariistuff.json similarity index 100% rename from data/developers/SicariusSicariiStuff.json rename to data/developers/sicariussicariistuff.json diff --git a/data/developers/SkyOrbis.json b/data/developers/skyorbis.json similarity index 100% rename from data/developers/SkyOrbis.json rename to data/developers/skyorbis.json diff --git a/data/developers/Skywork.json b/data/developers/skywork.json similarity index 98% rename from data/developers/Skywork.json rename to data/developers/skywork.json index 99e0c1c396995d96ed17d92a31186ed29c117119..74d24a368540017020e88ed938e51159414ee9bc 100644 --- a/data/developers/Skywork.json +++ b/data/developers/skywork.json @@ -33,16 +33,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.938, + "reward-bench/Score": 0.7576, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Safety": 0.9422, + "reward-bench/Reasoning": 0.9606, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, - "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Reasoning": 0.9606 + "reward-bench/Ties": 0.8261 } }, { diff --git a/data/developers/Solshine.json b/data/developers/solshine.json similarity index 100% rename from data/developers/Solshine.json rename to data/developers/solshine.json diff --git a/data/developers/Sorawiz.json b/data/developers/sorawiz.json similarity index 100% rename from data/developers/Sorawiz.json rename to data/developers/sorawiz.json diff --git a/data/developers/Sourjayon.json b/data/developers/sourjayon.json similarity index 100% rename from data/developers/Sourjayon.json rename to data/developers/sourjayon.json diff --git a/data/developers/SpaceYL.json b/data/developers/spaceyl.json similarity index 100% rename from data/developers/SpaceYL.json rename to data/developers/spaceyl.json diff --git a/data/developers/Spestly.json b/data/developers/spestly.json similarity index 100% rename from data/developers/Spestly.json rename to data/developers/spestly.json diff --git a/data/developers/spow12.json b/data/developers/spow12.json index 39b5b7b70e78310868166d1a8d2f2ae80dc77b8f..fbe5ef8e79c87a1486a23e55dcad74bc6e42909c 100644 --- a/data/developers/spow12.json +++ b/data/developers/spow12.json @@ -49,12 +49,12 @@ "developer": "spow12", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6511, - "hfopenllm_v2/BBH": 0.5926, - "hfopenllm_v2/MATH Level 5": 0.1858, - "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/IFEval": 0.6517, + "hfopenllm_v2/BBH": 0.5908, + "hfopenllm_v2/MATH Level 5": 0.2032, + "hfopenllm_v2/GPQA": 0.3238, "hfopenllm_v2/MUSR": 0.3842, - "hfopenllm_v2/MMLU-PRO": 0.3836 + "hfopenllm_v2/MMLU-PRO": 0.3812 } } ] diff --git a/data/developers/Stark2008.json b/data/developers/stark2008.json similarity index 100% rename from data/developers/Stark2008.json rename to data/developers/stark2008.json diff --git a/data/developers/Steelskull.json b/data/developers/steelskull.json similarity index 100% rename from data/developers/Steelskull.json rename to data/developers/steelskull.json diff --git a/data/developers/StelleX.json b/data/developers/stellex.json similarity index 100% rename from data/developers/StelleX.json rename to data/developers/stellex.json diff --git a/data/developers/SultanR.json b/data/developers/sultanr.json similarity index 100% rename from data/developers/SultanR.json rename to data/developers/sultanr.json diff --git a/data/developers/Supichi.json b/data/developers/supichi.json similarity index 100% rename from data/developers/Supichi.json rename to data/developers/supichi.json diff --git a/data/developers/Svak.json b/data/developers/svak.json similarity index 100% rename from data/developers/Svak.json rename to data/developers/svak.json diff --git a/data/developers/Syed-Hasan-8503.json b/data/developers/syed-hasan-8503.json similarity index 100% rename from data/developers/Syed-Hasan-8503.json rename to data/developers/syed-hasan-8503.json diff --git a/data/developers/T145.json b/data/developers/t145.json similarity index 100% rename from data/developers/T145.json rename to data/developers/t145.json diff --git a/data/developers/Tarek07.json b/data/developers/tarek07.json similarity index 100% rename from data/developers/Tarek07.json rename to data/developers/tarek07.json diff --git a/data/developers/TeeZee.json b/data/developers/teezee.json similarity index 100% rename from data/developers/TeeZee.json rename to data/developers/teezee.json diff --git a/data/developers/Telugu-LLM-Labs.json b/data/developers/telugu-llm-labs.json similarity index 100% rename from data/developers/Telugu-LLM-Labs.json rename to data/developers/telugu-llm-labs.json diff --git a/data/developers/TencentARC.json b/data/developers/tencentarc.json similarity index 100% rename from data/developers/TencentARC.json rename to data/developers/tencentarc.json diff --git a/data/developers/TheDrummer.json b/data/developers/thedrummer.json similarity index 100% rename from data/developers/TheDrummer.json rename to data/developers/thedrummer.json diff --git a/data/developers/TheDrunkenSnail.json b/data/developers/thedrunkensnail.json similarity index 100% rename from data/developers/TheDrunkenSnail.json rename to data/developers/thedrunkensnail.json diff --git a/data/developers/TheHierophant.json b/data/developers/thehierophant.json similarity index 100% rename from data/developers/TheHierophant.json rename to data/developers/thehierophant.json diff --git a/data/developers/TheTsar1209.json b/data/developers/thetsar1209.json similarity index 100% rename from data/developers/TheTsar1209.json rename to data/developers/thetsar1209.json diff --git a/data/developers/THUDM.json b/data/developers/thudm.json similarity index 100% rename from data/developers/THUDM.json rename to data/developers/thudm.json diff --git a/data/developers/TIGER-Lab.json b/data/developers/tiger-lab.json similarity index 100% rename from data/developers/TIGER-Lab.json rename to data/developers/tiger-lab.json diff --git a/data/developers/Tijmen2.json b/data/developers/tijmen2.json similarity index 100% rename from data/developers/Tijmen2.json rename to data/developers/tijmen2.json diff --git a/data/developers/TinyLlama.json b/data/developers/tinyllama.json similarity index 100% rename from data/developers/TinyLlama.json rename to data/developers/tinyllama.json diff --git a/data/developers/ToastyPigeon.json b/data/developers/toastypigeon.json similarity index 100% rename from data/developers/ToastyPigeon.json rename to data/developers/toastypigeon.json diff --git a/data/developers/Trappu.json b/data/developers/trappu.json similarity index 100% rename from data/developers/Trappu.json rename to data/developers/trappu.json diff --git a/data/developers/Tremontaine.json b/data/developers/tremontaine.json similarity index 100% rename from data/developers/Tremontaine.json rename to data/developers/tremontaine.json diff --git a/data/developers/Triangle104.json b/data/developers/triangle104.json similarity index 100% rename from data/developers/Triangle104.json rename to data/developers/triangle104.json diff --git a/data/developers/Tsunami-th.json b/data/developers/tsunami-th.json similarity index 100% rename from data/developers/Tsunami-th.json rename to data/developers/tsunami-th.json diff --git a/data/developers/TTTXXX01.json b/data/developers/tttxxx01.json similarity index 100% rename from data/developers/TTTXXX01.json rename to data/developers/tttxxx01.json diff --git a/data/developers/UCLA-AGI.json b/data/developers/ucla-agi.json similarity index 95% rename from data/developers/UCLA-AGI.json rename to data/developers/ucla-agi.json index 75dd11b6fe190b43556ce10dc96eff4392988245..93fb82e5b4d9ef216b964e7c543f4f7a063aa502 100644 --- a/data/developers/UCLA-AGI.json +++ b/data/developers/ucla-agi.json @@ -77,12 +77,12 @@ "developer": "UCLA-AGI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6834, - "hfopenllm_v2/BBH": 0.508, - "hfopenllm_v2/MATH Level 5": 0.0959, + "hfopenllm_v2/IFEval": 0.6703, + "hfopenllm_v2/BBH": 0.5076, + "hfopenllm_v2/MATH Level 5": 0.0718, "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3661, - "hfopenllm_v2/MMLU-PRO": 0.3644 + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.3658 } }, { diff --git a/data/developers/UKzExecution.json b/data/developers/ukzexecution.json similarity index 100% rename from data/developers/UKzExecution.json rename to data/developers/ukzexecution.json diff --git a/data/developers/Unbabel.json b/data/developers/unbabel.json similarity index 100% rename from data/developers/Unbabel.json rename to data/developers/unbabel.json diff --git a/data/developers/Undi95.json b/data/developers/undi95.json similarity index 100% rename from data/developers/Undi95.json rename to data/developers/undi95.json diff --git a/data/developers/unknown.json b/data/developers/unknown.json index 3c55382e79aaa4313dd242546596ab28b509e7ca..caadd81b81e09e8905405fc2663af88694315b8a 100644 --- a/data/developers/unknown.json +++ b/data/developers/unknown.json @@ -65,6 +65,24 @@ "reward-bench/Reasoning": 0.7575 } }, + { + "id": "meta-llama/Meta-Llama-3.1-8B", + "name": "Meta Llama 3.1 8B", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 27.04 + } + }, + { + "id": "meta-llama/Meta-Llama-3.1-8B-Instruct", + "name": "Meta Llama 3.1 8B Instruct", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 30.23 + } + }, { "id": "unknown/aya-expanse-32b", "name": "aya-expanse-32b", @@ -145,6 +163,15 @@ "global-mmlu-lite/Chinese": 0.89, "global-mmlu-lite/Burmese": 0.8725 } + }, + { + "id": "utter-project/EuroLLM-9B", + "name": "EuroLLM 9B", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 25.87 + } } ] } \ No newline at end of file diff --git a/data/developers/V3N0M.json b/data/developers/v3n0m.json similarity index 100% rename from data/developers/V3N0M.json rename to data/developers/v3n0m.json diff --git a/data/developers/VAGOsolutions.json b/data/developers/vagosolutions.json similarity index 100% rename from data/developers/VAGOsolutions.json rename to data/developers/vagosolutions.json diff --git a/data/developers/ValiantLabs.json b/data/developers/valiantlabs.json similarity index 95% rename from data/developers/ValiantLabs.json rename to data/developers/valiantlabs.json index 892d28b9bdd00084b849ccf11411be9ae5f19a1e..bf9fb4afa6f318f8db15eb852084df0656489446 100644 --- a/data/developers/ValiantLabs.json +++ b/data/developers/valiantlabs.json @@ -105,12 +105,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6496, - "hfopenllm_v2/BBH": 0.4774, - "hfopenllm_v2/MATH Level 5": 0.0566, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.3909, - "hfopenllm_v2/MMLU-PRO": 0.3382 + "hfopenllm_v2/IFEval": 0.2678, + "hfopenllm_v2/BBH": 0.4429, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.302, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.2927 } }, { diff --git a/data/developers/Vikhrmodels.json b/data/developers/vikhrmodels.json similarity index 100% rename from data/developers/Vikhrmodels.json rename to data/developers/vikhrmodels.json diff --git a/data/developers/VIRNECT.json b/data/developers/virnect.json similarity index 100% rename from data/developers/VIRNECT.json rename to data/developers/virnect.json diff --git a/data/developers/weqweasdas.json b/data/developers/weqweasdas.json index 93825681001579e2a4406f81036de228a63594a2..41634316953e64adcf8dfec1bf82c885064f6beb 100644 --- a/data/developers/weqweasdas.json +++ b/data/developers/weqweasdas.json @@ -59,17 +59,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7982, + "reward-bench/Score": 0.596, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Safety": 0.6911, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753 + "reward-bench/Ties": 0.6226 } }, { diff --git a/data/developers/Weyaxi.json b/data/developers/weyaxi.json similarity index 100% rename from data/developers/Weyaxi.json rename to data/developers/weyaxi.json diff --git a/data/developers/WizardLMTeam.json b/data/developers/wizardlmteam.json similarity index 100% rename from data/developers/WizardLMTeam.json rename to data/developers/wizardlmteam.json diff --git a/data/developers/Wladastic.json b/data/developers/wladastic.json similarity index 100% rename from data/developers/Wladastic.json rename to data/developers/wladastic.json diff --git a/data/developers/xAI.json b/data/developers/xAI.json deleted file mode 100644 index a178d74d4ce2ff741ee7cfef87dacd26423b9076..0000000000000000000000000000000000000000 --- a/data/developers/xAI.json +++ /dev/null @@ -1,23 +0,0 @@ -{ - "developer": "xAI", - "models": [ - { - "id": "xai/grok-4", - "name": "Grok 4", - "developer": "xAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.1 - } - }, - { - "id": "xai/grok-code-fast-1", - "name": "Grok Code Fast 1", - "developer": "xAI", - "evaluator_relationship": null, - "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 14.2 - } - } - ] -} \ No newline at end of file diff --git a/data/developers/xai.json b/data/developers/xai.json index 9373b5609f40cbf6169d56916e046772b21e5127..f433d52c08a0e026e1495ce27730a08f3b883dad 100644 --- a/data/developers/xai.json +++ b/data/developers/xai.json @@ -1,5 +1,5 @@ { - "developer": "xai", + "developer": "xAI", "models": [ { "id": "xai/Grok 4", @@ -72,6 +72,15 @@ "helm_capabilities/Omni-MATH": 0.318 } }, + { + "id": "xai/grok-4", + "name": "Grok 4", + "developer": "xAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 25.4 + } + }, { "id": "xai/grok-4-0709", "name": "grok-4-0709", @@ -104,6 +113,15 @@ "helm_capabilities/WildBench": 0.797, "helm_capabilities/Omni-MATH": 0.603 } + }, + { + "id": "xai/grok-code-fast-1", + "name": "Grok Code Fast 1", + "developer": "xAI", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 25.8 + } } ] } \ No newline at end of file diff --git a/data/developers/Xclbr7.json b/data/developers/xclbr7.json similarity index 100% rename from data/developers/Xclbr7.json rename to data/developers/xclbr7.json diff --git a/data/developers/Xiaojian9992024.json b/data/developers/xiaojian9992024.json similarity index 100% rename from data/developers/Xiaojian9992024.json rename to data/developers/xiaojian9992024.json diff --git a/data/developers/Xkev.json b/data/developers/xkev.json similarity index 100% rename from data/developers/Xkev.json rename to data/developers/xkev.json diff --git a/data/developers/xMaulana.json b/data/developers/xmaulana.json similarity index 100% rename from data/developers/xMaulana.json rename to data/developers/xmaulana.json diff --git a/data/developers/xxx777xxxASD.json b/data/developers/xxx777xxxasd.json similarity index 100% rename from data/developers/xxx777xxxASD.json rename to data/developers/xxx777xxxasd.json diff --git a/data/developers/yam-peleg.json b/data/developers/yam-peleg.json index 3161f95274f1998cddb90c53a59dce4097dbe0fd..f415128b8e93253c12ae15436b68cf36d5f248bf 100644 --- a/data/developers/yam-peleg.json +++ b/data/developers/yam-peleg.json @@ -35,12 +35,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } } ] diff --git a/data/developers/Yash21.json b/data/developers/yash21.json similarity index 100% rename from data/developers/Yash21.json rename to data/developers/yash21.json diff --git a/data/developers/yifAI.json b/data/developers/yifai.json similarity index 100% rename from data/developers/yifAI.json rename to data/developers/yifai.json diff --git a/data/developers/Youlln.json b/data/developers/youlln.json similarity index 100% rename from data/developers/Youlln.json rename to data/developers/youlln.json diff --git a/data/developers/YoungPanda.json b/data/developers/youngpanda.json similarity index 100% rename from data/developers/YoungPanda.json rename to data/developers/youngpanda.json diff --git a/data/developers/YOYO-AI.json b/data/developers/yoyo-ai.json similarity index 100% rename from data/developers/YOYO-AI.json rename to data/developers/yoyo-ai.json diff --git a/data/developers/Yuma42.json b/data/developers/yuma42.json similarity index 100% rename from data/developers/Yuma42.json rename to data/developers/yuma42.json diff --git a/data/developers/Z-AI.json b/data/developers/z-ai.json similarity index 100% rename from data/developers/Z-AI.json rename to data/developers/z-ai.json diff --git a/data/developers/Z.AI.json b/data/developers/z.ai.json similarity index 63% rename from data/developers/Z.AI.json rename to data/developers/z.ai.json index ef7b8b09a0447d151ec048fad8785a407220b327..61c8b659f6eee02cc2d6daa2330e083662c9e675 100644 --- a/data/developers/Z.AI.json +++ b/data/developers/z.ai.json @@ -11,6 +11,15 @@ "livecodebenchpro/Medium Problems": 0.028169014084507043, "livecodebenchpro/Easy Problems": 0.1267605633802817 } + }, + { + "id": "zhipu-ai/glm-4.6", + "name": "GLM 4.6", + "developer": "Z.ai", + "evaluator_relationship": null, + "benchmark_scores": { + "terminal-bench-2.0/terminal-bench-2.0": 24.5 + } } ] } \ No newline at end of file diff --git a/data/developers/Z1-Coder.json b/data/developers/z1-coder.json similarity index 100% rename from data/developers/Z1-Coder.json rename to data/developers/z1-coder.json diff --git a/data/developers/ZeroXClem.json b/data/developers/zeroxclem.json similarity index 100% rename from data/developers/ZeroXClem.json rename to data/developers/zeroxclem.json diff --git a/data/developers/ZeusLabs.json b/data/developers/zeuslabs.json similarity index 100% rename from data/developers/ZeusLabs.json rename to data/developers/zeuslabs.json diff --git a/data/developers/ZhangShenao.json b/data/developers/zhangshenao.json similarity index 100% rename from data/developers/ZhangShenao.json rename to data/developers/zhangshenao.json diff --git a/data/developers/ZHLiu627.json b/data/developers/zhliu627.json similarity index 100% rename from data/developers/ZHLiu627.json rename to data/developers/zhliu627.json diff --git a/data/developers/ZiyiYe.json b/data/developers/ziyiye.json similarity index 100% rename from data/developers/ZiyiYe.json rename to data/developers/ziyiye.json diff --git a/data/models.json b/data/models.json index 8c41767f8aba7274867583a42ae301bdc9ee2ff6..e0a63d2fe5c1ad114b298407343dffcd46fa8132 100644 --- a/data/models.json +++ b/data/models.json @@ -1446,12 +1446,12 @@ "developer": "AtAndDev", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4511, - "hfopenllm_v2/BBH": 0.4275, - "hfopenllm_v2/MATH Level 5": 0.1473, - "hfopenllm_v2/GPQA": 0.2701, - "hfopenllm_v2/MUSR": 0.3623, - "hfopenllm_v2/MMLU-PRO": 0.2806 + "hfopenllm_v2/IFEval": 0.4605, + "hfopenllm_v2/BBH": 0.4258, + "hfopenllm_v2/MATH Level 5": 0.0748, + "hfopenllm_v2/GPQA": 0.2659, + "hfopenllm_v2/MUSR": 0.3636, + "hfopenllm_v2/MMLU-PRO": 0.2812 } }, { @@ -2354,17 +2354,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } }, { @@ -3935,12 +3935,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4383, - "hfopenllm_v2/BBH": 0.5034, - "hfopenllm_v2/MATH Level 5": 0.1443, + "hfopenllm_v2/IFEval": 0.4398, + "hfopenllm_v2/BBH": 0.5066, + "hfopenllm_v2/MATH Level 5": 0.1488, "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.4052, - "hfopenllm_v2/MMLU-PRO": 0.3778 + "hfopenllm_v2/MUSR": 0.4079, + "hfopenllm_v2/MMLU-PRO": 0.3804 } }, { @@ -4131,12 +4131,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3745, - "hfopenllm_v2/BBH": 0.6668, - "hfopenllm_v2/MATH Level 5": 0.4758, - "hfopenllm_v2/GPQA": 0.3943, - "hfopenllm_v2/MUSR": 0.4858, - "hfopenllm_v2/MMLU-PRO": 0.5593 + "hfopenllm_v2/IFEval": 0.4855, + "hfopenllm_v2/BBH": 0.6627, + "hfopenllm_v2/MATH Level 5": 0.4841, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4256, + "hfopenllm_v2/MMLU-PRO": 0.5542 } }, { @@ -4986,12 +4986,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { @@ -9788,12 +9788,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3472, - "hfopenllm_v2/BBH": 0.3268, - "hfopenllm_v2/MATH Level 5": 0.0891, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3262, - "hfopenllm_v2/MMLU-PRO": 0.1641 + "hfopenllm_v2/IFEval": 0.3417, + "hfopenllm_v2/BBH": 0.3292, + "hfopenllm_v2/MATH Level 5": 0.0023, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3249, + "hfopenllm_v2/MMLU-PRO": 0.1638 } }, { @@ -9914,12 +9914,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7598, - "hfopenllm_v2/BBH": 0.5107, - "hfopenllm_v2/MATH Level 5": 0.4237, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.4539, - "hfopenllm_v2/MMLU-PRO": 0.4012 + "hfopenllm_v2/IFEval": 0.7628, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2802, + "hfopenllm_v2/MUSR": 0.4579, + "hfopenllm_v2/MMLU-PRO": 0.4033 } }, { @@ -10082,12 +10082,12 @@ "developer": "Gunulhona", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.288, - "hfopenllm_v2/BBH": 0.5154, + "hfopenllm_v2/IFEval": 0.4441, + "hfopenllm_v2/BBH": 0.4863, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.3247, - "hfopenllm_v2/MUSR": 0.408, - "hfopenllm_v2/MMLU-PRO": 0.3817 + "hfopenllm_v2/GPQA": 0.307, + "hfopenllm_v2/MUSR": 0.3986, + "hfopenllm_v2/MMLU-PRO": 0.3098 } }, { @@ -10577,12 +10577,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0593, - "hfopenllm_v2/BBH": 0.3135, - "hfopenllm_v2/MATH Level 5": 0.0144, - "hfopenllm_v2/GPQA": 0.2341, - "hfopenllm_v2/MUSR": 0.3871, - "hfopenllm_v2/MMLU-PRO": 0.1092 + "hfopenllm_v2/IFEval": 0.2883, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.003, + "hfopenllm_v2/GPQA": 0.2357, + "hfopenllm_v2/MUSR": 0.3662, + "hfopenllm_v2/MMLU-PRO": 0.1115 } }, { @@ -10605,12 +10605,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } }, { @@ -10843,12 +10843,12 @@ "developer": "Isaak-Carter", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2553, - "hfopenllm_v2/BBH": 0.4725, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.3654, - "hfopenllm_v2/MMLU-PRO": 0.3316 + "hfopenllm_v2/IFEval": 0.2477, + "hfopenllm_v2/BBH": 0.4758, + "hfopenllm_v2/MATH Level 5": 0.0453, + "hfopenllm_v2/GPQA": 0.2911, + "hfopenllm_v2/MUSR": 0.3641, + "hfopenllm_v2/MMLU-PRO": 0.3292 } }, { @@ -16022,16 +16022,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9294, + "reward-bench/Score": 0.7394, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Safety": 0.9178, + "reward-bench/Reasoning": 0.9698, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, - "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Reasoning": 0.9698 + "reward-bench/Ties": 0.7653 } }, { @@ -16180,12 +16180,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4118, - "hfopenllm_v2/BBH": 0.4811, - "hfopenllm_v2/MATH Level 5": 0.034, - "hfopenllm_v2/GPQA": 0.2752, - "hfopenllm_v2/MUSR": 0.3047, - "hfopenllm_v2/MMLU-PRO": 0.3006 + "hfopenllm_v2/IFEval": 0.4027, + "hfopenllm_v2/BBH": 0.4789, + "hfopenllm_v2/MATH Level 5": 0.0461, + "hfopenllm_v2/GPQA": 0.2768, + "hfopenllm_v2/MUSR": 0.3087, + "hfopenllm_v2/MMLU-PRO": 0.3001 } }, { @@ -17481,16 +17481,16 @@ "developer": "NCSOFT", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.648, - "reward-bench/Chat": 0.9721, - "reward-bench/Chat Hard": 0.818, - "reward-bench/Safety": 0.7222, - "reward-bench/Reasoning": 0.9192, + "reward-bench/Score": 0.8942, "reward-bench/Factuality": 0.6084, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5191, + "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.9596, - "reward-bench/Ties": 0.6786 + "reward-bench/Ties": 0.6786, + "reward-bench/Chat": 0.9721, + "reward-bench/Chat Hard": 0.818, + "reward-bench/Reasoning": 0.9192 } }, { @@ -18395,17 +18395,17 @@ "developer": "Nexusflow", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8133, + "reward-bench/Score": 0.4553, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7556, + "reward-bench/Reasoning": 0.8845, + "reward-bench/Prior Sets (0.5 weight)": 0.7137, "reward-bench/Factuality": 0.4589, "reward-bench/Precise IF": 0.3187, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.877, "reward-bench/Focus": 0.4808, - "reward-bench/Ties": 0.1004, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8845, - "reward-bench/Prior Sets (0.5 weight)": 0.7137 + "reward-bench/Ties": 0.1004 } }, { @@ -19369,12 +19369,12 @@ "developer": "Omkar1102", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2254, - "hfopenllm_v2/BBH": 0.275, + "hfopenllm_v2/IFEval": 0.2148, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3762, - "hfopenllm_v2/MMLU-PRO": 0.1123 + "hfopenllm_v2/GPQA": 0.2508, + "hfopenllm_v2/MUSR": 0.3802, + "hfopenllm_v2/MMLU-PRO": 0.1126 } }, { @@ -19477,17 +19477,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.32, - "reward-bench/Chat": 0.8939, - "reward-bench/Chat Hard": 0.4518, - "reward-bench/Safety": 0.3667, - "reward-bench/Reasoning": 0.3855, - "reward-bench/Prior Sets (0.5 weight)": 0.5836, + "reward-bench/Score": 0.6126, "reward-bench/Factuality": 0.3853, "reward-bench/Precise IF": 0.2687, "reward-bench/Math": 0.5027, + "reward-bench/Safety": 0.7338, "reward-bench/Focus": 0.2768, - "reward-bench/Ties": 0.12 + "reward-bench/Ties": 0.12, + "reward-bench/Chat": 0.8939, + "reward-bench/Chat Hard": 0.4518, + "reward-bench/Reasoning": 0.3855, + "reward-bench/Prior Sets (0.5 weight)": 0.5836 } }, { @@ -20126,17 +20126,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3332, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Safety": 0.7589, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57, + "reward-bench/Score": 0.5798, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, + "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57 } }, { @@ -20145,17 +20145,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4727, + "reward-bench/Score": 0.1606, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Safety": 0.1422, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, - "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993 + "reward-bench/Ties": -0.01 } }, { @@ -20164,17 +20164,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3326, - "reward-bench/Chat": 0.5726, - "reward-bench/Chat Hard": 0.4561, - "reward-bench/Safety": 0.7356, - "reward-bench/Reasoning": 0.6211, - "reward-bench/Prior Sets (0.5 weight)": 0.5397, + "reward-bench/Score": 0.5957, "reward-bench/Factuality": 0.3789, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3333, + "reward-bench/Safety": 0.7608, "reward-bench/Focus": 0.2828, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.5726, + "reward-bench/Chat Hard": 0.4561, + "reward-bench/Reasoning": 0.6211, + "reward-bench/Prior Sets (0.5 weight)": 0.5397 } }, { @@ -20663,12 +20663,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6718, - "hfopenllm_v2/BBH": 0.6891, - "hfopenllm_v2/MATH Level 5": 0.4985, - "hfopenllm_v2/GPQA": 0.3339, - "hfopenllm_v2/MUSR": 0.4323, - "hfopenllm_v2/MMLU-PRO": 0.5408 + "hfopenllm_v2/IFEval": 0.6654, + "hfopenllm_v2/BBH": 0.6901, + "hfopenllm_v2/MATH Level 5": 0.4698, + "hfopenllm_v2/GPQA": 0.3331, + "hfopenllm_v2/MUSR": 0.431, + "hfopenllm_v2/MMLU-PRO": 0.5426 } }, { @@ -22085,12 +22085,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3071, - "hfopenllm_v2/BBH": 0.3341, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3329, - "hfopenllm_v2/MMLU-PRO": 0.1697 + "hfopenllm_v2/IFEval": 0.3153, + "hfopenllm_v2/BBH": 0.3322, + "hfopenllm_v2/MATH Level 5": 0.1035, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3342, + "hfopenllm_v2/MMLU-PRO": 0.172 } }, { @@ -22258,7 +22258,8 @@ "hfopenllm_v2/MATH Level 5": 0.2508, "hfopenllm_v2/GPQA": 0.3247, "hfopenllm_v2/MUSR": 0.4424, - "hfopenllm_v2/MMLU-PRO": 0.4365 + "hfopenllm_v2/MMLU-PRO": 0.4365, + "la_leaderboard/la_leaderboard": 27.61 } }, { @@ -22365,12 +22366,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6101, - "hfopenllm_v2/BBH": 0.5008, - "hfopenllm_v2/MATH Level 5": 0.3716, - "hfopenllm_v2/GPQA": 0.2919, - "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3352 + "hfopenllm_v2/IFEval": 0.6147, + "hfopenllm_v2/BBH": 0.4999, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2936, + "hfopenllm_v2/MUSR": 0.4099, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { @@ -22692,16 +22693,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8839, + "reward-bench/Score": 0.5966, + "reward-bench/Chat": 0.9302, + "reward-bench/Chat Hard": 0.7719, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.912, "reward-bench/Factuality": 0.5305, "reward-bench/Precise IF": 0.3125, "reward-bench/Math": 0.5902, - "reward-bench/Safety": 0.9216, "reward-bench/Focus": 0.7455, - "reward-bench/Ties": 0.4788, - "reward-bench/Chat": 0.9302, - "reward-bench/Chat Hard": 0.7719, - "reward-bench/Reasoning": 0.912 + "reward-bench/Ties": 0.4788 } }, { @@ -22886,12 +22887,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { @@ -24576,16 +24577,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } }, { @@ -25110,16 +25111,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.938, + "reward-bench/Score": 0.7576, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Safety": 0.9422, + "reward-bench/Reasoning": 0.9606, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, - "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Reasoning": 0.9606 + "reward-bench/Ties": 0.8261 } }, { @@ -28236,12 +28237,12 @@ "developer": "UCLA-AGI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6834, - "hfopenllm_v2/BBH": 0.508, - "hfopenllm_v2/MATH Level 5": 0.0959, + "hfopenllm_v2/IFEval": 0.6703, + "hfopenllm_v2/BBH": 0.5076, + "hfopenllm_v2/MATH Level 5": 0.0718, "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3661, - "hfopenllm_v2/MMLU-PRO": 0.3644 + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.3658 } }, { @@ -28740,12 +28741,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6496, - "hfopenllm_v2/BBH": 0.4774, - "hfopenllm_v2/MATH Level 5": 0.0566, - "hfopenllm_v2/GPQA": 0.3104, - "hfopenllm_v2/MUSR": 0.3909, - "hfopenllm_v2/MMLU-PRO": 0.3382 + "hfopenllm_v2/IFEval": 0.2678, + "hfopenllm_v2/BBH": 0.4429, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.302, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.2927 } }, { @@ -30251,12 +30252,12 @@ "developer": "abhishek", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1952, - "hfopenllm_v2/BBH": 0.3127, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3584, - "hfopenllm_v2/MMLU-PRO": 0.1144 + "hfopenllm_v2/IFEval": 0.1957, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.365, + "hfopenllm_v2/MMLU-PRO": 0.1151 } }, { @@ -30553,10 +30554,10 @@ "developer": "ai2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6895, + "reward-bench/Score": 0.7008, "reward-bench/Chat": 0.9385, - "reward-bench/Chat Hard": 0.3706, - "reward-bench/Safety": 0.7595 + "reward-bench/Chat Hard": 0.3882, + "reward-bench/Safety": 0.7757 } }, { @@ -31197,7 +31198,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.9 + "terminal-bench-2.0/terminal-bench-2.0": 27.2 } }, { @@ -31288,17 +31289,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9021, + "reward-bench/Score": 0.7606, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.8355, + "reward-bench/Safety": 0.8844, + "reward-bench/Reasoning": 0.8969, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8126, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9095, "reward-bench/Focus": 0.8646, - "reward-bench/Ties": 0.8835, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.8355, - "reward-bench/Reasoning": 0.8969, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8835 } }, { @@ -31307,17 +31308,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.649, - "reward-bench/Chat": 0.933, - "reward-bench/Chat Hard": 0.7785, - "reward-bench/Safety": 0.8267, - "reward-bench/Reasoning": 0.7886, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8463, "reward-bench/Factuality": 0.72, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.8851, "reward-bench/Focus": 0.8323, - "reward-bench/Ties": 0.5406 + "reward-bench/Ties": 0.5406, + "reward-bench/Chat": 0.933, + "reward-bench/Chat Hard": 0.7785, + "reward-bench/Reasoning": 0.7886, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -31326,17 +31327,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8885, + "reward-bench/Score": 0.7285, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.8158, + "reward-bench/Safety": 0.8956, + "reward-bench/Reasoning": 0.887, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7432, "reward-bench/Precise IF": 0.4437, "reward-bench/Math": 0.6175, - "reward-bench/Safety": 0.8932, "reward-bench/Focus": 0.9071, - "reward-bench/Ties": 0.7638, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.8158, - "reward-bench/Reasoning": 0.887, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.7638 } }, { @@ -31345,12 +31346,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8291, - "hfopenllm_v2/BBH": 0.6164, - "hfopenllm_v2/MATH Level 5": 0.4502, + "hfopenllm_v2/IFEval": 0.8379, + "hfopenllm_v2/BBH": 0.6157, + "hfopenllm_v2/MATH Level 5": 0.3829, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4948, - "hfopenllm_v2/MMLU-PRO": 0.4645 + "hfopenllm_v2/MUSR": 0.4988, + "hfopenllm_v2/MMLU-PRO": 0.4656 } }, { @@ -31387,17 +31388,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8892, + "reward-bench/Score": 0.722, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Safety": 0.8689, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, - "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8308 } }, { @@ -31406,12 +31407,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8255, - "hfopenllm_v2/BBH": 0.4061, - "hfopenllm_v2/MATH Level 5": 0.2115, - "hfopenllm_v2/GPQA": 0.297, + "hfopenllm_v2/IFEval": 0.8267, + "hfopenllm_v2/BBH": 0.405, + "hfopenllm_v2/MATH Level 5": 0.1964, + "hfopenllm_v2/GPQA": 0.2987, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2821 + "hfopenllm_v2/MMLU-PRO": 0.2827 } }, { @@ -35662,8 +35663,6 @@ "developer": "anthropic", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.478, - "ace/Gaming Score": 0.391, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.34, "apex-agents/Overall Mean Score": 0.348, @@ -35671,6 +35670,8 @@ "apex-agents/Management Consulting Pass@1": 0.132, "apex-agents/Corporate Law Pass@1": 0.202, "apex-agents/Corporate Lawyer Mean Score": 0.471, + "ace/Overall Score": 0.478, + "ace/Gaming Score": 0.391, "apex-v1/Medicine (MD) Score": 0.65 } }, @@ -36202,7 +36203,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.5 + "terminal-bench-2.0/terminal-bench-2.0": 29.8 } }, { @@ -36327,12 +36328,12 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.66, - "browsecompplus/browsecompplus": 0.49, - "swe-bench/swe-bench": 0.65, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, - "tau-bench-2_retail/tau-bench-2/retail": 0.85, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.58 + "appworld_test_normal/appworld/test_normal": 0.61, + "browsecompplus/browsecompplus": 0.61, + "swe-bench/swe-bench": 0.6061, + "tau-bench-2_airline/tau-bench-2/airline": 0.74, + "tau-bench-2_retail/tau-bench-2/retail": 0.78, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 } }, { @@ -36341,7 +36342,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 38.0 + "terminal-bench-2.0/terminal-bench-2.0": 36.9 } }, { @@ -36350,7 +36351,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 54.3 + "terminal-bench-2.0/terminal-bench-2.0": 52.1 } }, { @@ -36359,7 +36360,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 69.9 + "terminal-bench-2.0/terminal-bench-2.0": 62.9 } }, { @@ -36433,7 +36434,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.6 + "terminal-bench-2.0/terminal-bench-2.0": 46.5 } }, { @@ -38412,12 +38413,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4652, - "hfopenllm_v2/BBH": 0.4531, - "hfopenllm_v2/MATH Level 5": 0.1284, - "hfopenllm_v2/GPQA": 0.2643, - "hfopenllm_v2/MUSR": 0.3394, - "hfopenllm_v2/MMLU-PRO": 0.3152 + "hfopenllm_v2/IFEval": 0.1775, + "hfopenllm_v2/BBH": 0.295, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.3647, + "hfopenllm_v2/MMLU-PRO": 0.1049 } }, { @@ -39809,12 +39810,12 @@ "developer": "cognitivecomputations", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4124, - "hfopenllm_v2/BBH": 0.6383, - "hfopenllm_v2/MATH Level 5": 0.182, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4349, - "hfopenllm_v2/MMLU-PRO": 0.4525 + "hfopenllm_v2/IFEval": 0.3613, + "hfopenllm_v2/BBH": 0.6123, + "hfopenllm_v2/MATH Level 5": 0.1239, + "hfopenllm_v2/GPQA": 0.328, + "hfopenllm_v2/MUSR": 0.4112, + "hfopenllm_v2/MMLU-PRO": 0.4494 } }, { @@ -40360,12 +40361,12 @@ "developer": "cpayne1303", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1916, - "hfopenllm_v2/BBH": 0.2977, - "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/IFEval": 0.1949, + "hfopenllm_v2/BBH": 0.2965, + "hfopenllm_v2/MATH Level 5": 0.0045, "hfopenllm_v2/GPQA": 0.2685, - "hfopenllm_v2/MUSR": 0.3872, - "hfopenllm_v2/MMLU-PRO": 0.1132 + "hfopenllm_v2/MUSR": 0.3885, + "hfopenllm_v2/MMLU-PRO": 0.1111 } }, { @@ -42553,12 +42554,12 @@ "developer": "fblgit", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4503, - "hfopenllm_v2/BBH": 0.7035, - "hfopenllm_v2/MATH Level 5": 0.3943, - "hfopenllm_v2/GPQA": 0.401, - "hfopenllm_v2/MUSR": 0.5021, - "hfopenllm_v2/MMLU-PRO": 0.5911 + "hfopenllm_v2/IFEval": 0.5181, + "hfopenllm_v2/BBH": 0.7033, + "hfopenllm_v2/MATH Level 5": 0.4947, + "hfopenllm_v2/GPQA": 0.3826, + "hfopenllm_v2/MUSR": 0.5008, + "hfopenllm_v2/MMLU-PRO": 0.5915 } }, { @@ -43705,6 +43706,7 @@ "developer": "google", "evaluator_relationship": null, "benchmark_scores": { + "ace/Gaming Score": 0.415, "apex-agents/Overall Pass@1": 0.24, "apex-agents/Overall Pass@8": 0.367, "apex-agents/Overall Mean Score": 0.395, @@ -43712,7 +43714,6 @@ "apex-agents/Management Consulting Pass@1": 0.193, "apex-agents/Corporate Law Pass@1": 0.259, "apex-agents/Corporate Lawyer Mean Score": 0.524, - "ace/Gaming Score": 0.415, "apex-v1/Overall Score": 0.64, "apex-v1/Consulting Score": 0.64 } @@ -43891,12 +43892,12 @@ "developer": "google", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2207, - "hfopenllm_v2/BBH": 0.4537, - "hfopenllm_v2/MATH Level 5": 0.0008, - "hfopenllm_v2/GPQA": 0.2458, - "hfopenllm_v2/MUSR": 0.422, - "hfopenllm_v2/MMLU-PRO": 0.2142 + "hfopenllm_v2/IFEval": 0.2237, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.0076, + "hfopenllm_v2/GPQA": 0.2525, + "hfopenllm_v2/MUSR": 0.4181, + "hfopenllm_v2/MMLU-PRO": 0.2147 } }, { @@ -44616,7 +44617,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.2 + "terminal-bench-2.0/terminal-bench-2.0": 56.9 } }, { @@ -44625,7 +44626,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.505, + "appworld_test_normal/appworld/test_normal": 0.36, "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, @@ -44646,8 +44647,8 @@ "global-mmlu-lite/Yoruba": 0.9425, "global-mmlu-lite/Chinese": 0.9475, "global-mmlu-lite/Burmese": 0.9425, - "swe-bench/swe-bench": 0.7234, - "tau-bench-2_airline/tau-bench-2/airline": 0.68, + "swe-bench/swe-bench": 0.71, + "tau-bench-2_airline/tau-bench-2/airline": 0.7, "tau-bench-2_retail/tau-bench-2/retail": 0.7805, "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 } @@ -44658,7 +44659,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 78.4 + "terminal-bench-2.0/terminal-bench-2.0": 74.8 } }, { @@ -44881,7 +44882,8 @@ "hfopenllm_v2/MATH Level 5": 0.1949, "hfopenllm_v2/GPQA": 0.3607, "hfopenllm_v2/MUSR": 0.4073, - "hfopenllm_v2/MMLU-PRO": 0.3875 + "hfopenllm_v2/MMLU-PRO": 0.3875, + "la_leaderboard/la_leaderboard": 33.62 } }, { @@ -45868,17 +45870,17 @@ "developer": "hendrydong", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7847, + "reward-bench/Score": 0.5851, + "reward-bench/Chat": 0.9832, + "reward-bench/Chat Hard": 0.5789, + "reward-bench/Safety": 0.6956, + "reward-bench/Reasoning": 0.7434, + "reward-bench/Prior Sets (0.5 weight)": 0.7508, "reward-bench/Factuality": 0.5779, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.6011, - "reward-bench/Safety": 0.85, "reward-bench/Focus": 0.6747, - "reward-bench/Ties": 0.5988, - "reward-bench/Chat": 0.9832, - "reward-bench/Chat Hard": 0.5789, - "reward-bench/Reasoning": 0.7434, - "reward-bench/Prior Sets (0.5 weight)": 0.7508 + "reward-bench/Ties": 0.5988 } }, { @@ -48000,16 +48002,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9511, + "reward-bench/Score": 0.7648, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Safety": 0.9644, + "reward-bench/Reasoning": 0.9912, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Reasoning": 0.9912 + "reward-bench/Ties": 0.8622 } }, { @@ -48074,16 +48076,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -48092,16 +48094,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5628, - "reward-bench/Chat": 0.9888, - "reward-bench/Chat Hard": 0.7654, - "reward-bench/Safety": 0.6111, - "reward-bench/Reasoning": 0.9576, + "reward-bench/Score": 0.9016, "reward-bench/Factuality": 0.5558, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.5738, + "reward-bench/Safety": 0.8946, "reward-bench/Focus": 0.7253, - "reward-bench/Ties": 0.5483 + "reward-bench/Ties": 0.5483, + "reward-bench/Chat": 0.9888, + "reward-bench/Chat Hard": 0.7654, + "reward-bench/Reasoning": 0.9576 } }, { @@ -48436,12 +48438,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4345, - "hfopenllm_v2/BBH": 0.5419, - "hfopenllm_v2/MATH Level 5": 0.1292, - "hfopenllm_v2/GPQA": 0.3087, + "hfopenllm_v2/IFEval": 0.4418, + "hfopenllm_v2/BBH": 0.5406, + "hfopenllm_v2/MATH Level 5": 0.1352, + "hfopenllm_v2/GPQA": 0.3062, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.3854 + "hfopenllm_v2/MMLU-PRO": 0.386 } }, { @@ -55271,6 +55273,24 @@ "reward-bench/Reasoning": 0.828 } }, + { + "id": "meta-llama/Meta-Llama-3.1-8B", + "name": "Meta Llama 3.1 8B", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 27.04 + } + }, + { + "id": "meta-llama/Meta-Llama-3.1-8B-Instruct", + "name": "Meta Llama 3.1 8B Instruct", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 30.23 + } + }, { "id": "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", "name": "meta-llama/Meta-Llama-3.1-8B-Instruct-Turbo", @@ -56603,12 +56623,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0585, - "hfopenllm_v2/BBH": 0.6691, - "hfopenllm_v2/MATH Level 5": 0.3165, - "hfopenllm_v2/GPQA": 0.406, + "hfopenllm_v2/IFEval": 0.0488, + "hfopenllm_v2/BBH": 0.6703, + "hfopenllm_v2/MATH Level 5": 0.2787, + "hfopenllm_v2/GPQA": 0.401, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5287 + "hfopenllm_v2/MMLU-PRO": 0.5295 } }, { @@ -56729,12 +56749,12 @@ "developer": "migtissera", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.443, - "hfopenllm_v2/BBH": 0.5706, - "hfopenllm_v2/MATH Level 5": 0.0869, - "hfopenllm_v2/GPQA": 0.3079, - "hfopenllm_v2/MUSR": 0.4031, - "hfopenllm_v2/MMLU-PRO": 0.3354 + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5686, + "hfopenllm_v2/MATH Level 5": 0.0838, + "hfopenllm_v2/GPQA": 0.3003, + "hfopenllm_v2/MUSR": 0.4045, + "hfopenllm_v2/MMLU-PRO": 0.334 } }, { @@ -56789,7 +56809,7 @@ "developer": "MiniMax", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 29.2 + "terminal-bench-2.0/terminal-bench-2.0": 36.6 } }, { @@ -57102,12 +57122,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2326, - "hfopenllm_v2/BBH": 0.5098, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.3205, - "hfopenllm_v2/MUSR": 0.4413, - "hfopenllm_v2/MMLU-PRO": 0.3871 + "hfopenllm_v2/IFEval": 0.2415, + "hfopenllm_v2/BBH": 0.5087, + "hfopenllm_v2/MATH Level 5": 0.102, + "hfopenllm_v2/GPQA": 0.3138, + "hfopenllm_v2/MUSR": 0.4321, + "hfopenllm_v2/MMLU-PRO": 0.385 } }, { @@ -57912,12 +57932,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4162, - "hfopenllm_v2/BBH": 0.5124, - "hfopenllm_v2/MATH Level 5": 0.0853, - "hfopenllm_v2/GPQA": 0.3029, - "hfopenllm_v2/MUSR": 0.415, - "hfopenllm_v2/MMLU-PRO": 0.3802 + "hfopenllm_v2/IFEval": 0.7561, + "hfopenllm_v2/BBH": 0.5111, + "hfopenllm_v2/MATH Level 5": 0.0906, + "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3841 } }, { @@ -58345,7 +58365,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 50.1 + "terminal-bench-2.0/terminal-bench-2.0": 72.4 } }, { @@ -60229,16 +60249,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9444, + "reward-bench/Score": 0.7667, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Safety": 0.9578, + "reward-bench/Reasoning": 0.9826, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Reasoning": 0.9826 + "reward-bench/Ties": 0.8321 } }, { @@ -60273,16 +60293,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9314, + "reward-bench/Score": 0.7074, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Safety": 0.9467, + "reward-bench/Reasoning": 0.9677, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, - "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Reasoning": 0.9677 + "reward-bench/Ties": 0.7234 } }, { @@ -60305,12 +60325,12 @@ "developer": "nisten", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3914, - "hfopenllm_v2/BBH": 0.6591, - "hfopenllm_v2/MATH Level 5": 0.3044, - "hfopenllm_v2/GPQA": 0.3591, - "hfopenllm_v2/MUSR": 0.4681, - "hfopenllm_v2/MMLU-PRO": 0.5611 + "hfopenllm_v2/IFEval": 0.3799, + "hfopenllm_v2/BBH": 0.6647, + "hfopenllm_v2/MATH Level 5": 0.3406, + "hfopenllm_v2/GPQA": 0.4035, + "hfopenllm_v2/MUSR": 0.494, + "hfopenllm_v2/MMLU-PRO": 0.5731 } }, { @@ -61240,12 +61260,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { @@ -61436,12 +61456,12 @@ "developer": "oopere", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2164, - "hfopenllm_v2/BBH": 0.3169, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2584, + "hfopenllm_v2/IFEval": 0.2119, + "hfopenllm_v2/BBH": 0.3156, + "hfopenllm_v2/MATH Level 5": 0.0181, + "hfopenllm_v2/GPQA": 0.2567, "hfopenllm_v2/MUSR": 0.3832, - "hfopenllm_v2/MMLU-PRO": 0.1134 + "hfopenllm_v2/MMLU-PRO": 0.113 } }, { @@ -62586,16 +62606,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 } }, { @@ -62658,7 +62678,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 31.9 + "terminal-bench-2.0/terminal-bench-2.0": 29.2 } }, { @@ -62681,7 +62701,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 7.0 + "terminal-bench-2.0/terminal-bench-2.0": 7.9 } }, { @@ -62713,7 +62733,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.9 + "terminal-bench-2.0/terminal-bench-2.0": 53.5 } }, { @@ -62749,15 +62769,15 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.071, + "appworld_test_normal/appworld/test_normal": 0.22, "browsecompplus/browsecompplus": 0.46, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, "swe-bench/swe-bench": 0.5253, - "tau-bench-2_airline/tau-bench-2/airline": 0.54, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 + "tau-bench-2_airline/tau-bench-2/airline": 0.6, + "tau-bench-2_retail/tau-bench-2/retail": 0.51, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.55 } }, { @@ -62793,7 +62813,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.11267605633802817, "livecodebenchpro/Easy Problems": 0.6619718309859155, - "terminal-bench-2.0/terminal-bench-2.0": 18.7 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } }, { @@ -62914,9 +62934,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.0143, - "livecodebenchpro/Medium Problems": 0.2923, - "livecodebenchpro/Easy Problems": 0.8571 + "livecodebenchpro/Hard Problems": 0.014084507042253521, + "livecodebenchpro/Medium Problems": 0.30985915492957744, + "livecodebenchpro/Easy Problems": 0.8873239436619719 } }, { @@ -63074,17 +63094,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5806, - "reward-bench/Chat": 0.9804, - "reward-bench/Chat Hard": 0.6557, - "reward-bench/Safety": 0.6267, - "reward-bench/Reasoning": 0.8633, - "reward-bench/Prior Sets (0.5 weight)": 0.7172, + "reward-bench/Score": 0.8159, "reward-bench/Factuality": 0.6, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5683, + "reward-bench/Safety": 0.8135, "reward-bench/Focus": 0.7475, - "reward-bench/Ties": 0.5972 + "reward-bench/Ties": 0.5972, + "reward-bench/Chat": 0.9804, + "reward-bench/Chat Hard": 0.6557, + "reward-bench/Reasoning": 0.8633, + "reward-bench/Prior Sets (0.5 weight)": 0.7172 } }, { @@ -63121,17 +63141,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6903, + "reward-bench/Score": 0.4683, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Safety": 0.5089, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, - "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294 + "reward-bench/Ties": 0.3036 } }, { @@ -66256,12 +66276,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2401, - "hfopenllm_v2/BBH": 0.4622, - "hfopenllm_v2/MATH Level 5": 0.0725, - "hfopenllm_v2/GPQA": 0.2609, - "hfopenllm_v2/MUSR": 0.3703, - "hfopenllm_v2/MMLU-PRO": 0.2379 + "hfopenllm_v2/IFEval": 0.2358, + "hfopenllm_v2/BBH": 0.4612, + "hfopenllm_v2/MATH Level 5": 0.0642, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3717, + "hfopenllm_v2/MMLU-PRO": 0.2382 } }, { @@ -66270,12 +66290,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { @@ -67134,12 +67154,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2854, - "hfopenllm_v2/BBH": 0.5984, - "hfopenllm_v2/MATH Level 5": 0.1005, - "hfopenllm_v2/GPQA": 0.3297, - "hfopenllm_v2/MUSR": 0.4607, - "hfopenllm_v2/MMLU-PRO": 0.4162 + "hfopenllm_v2/IFEval": 0.7649, + "hfopenllm_v2/BBH": 0.5974, + "hfopenllm_v2/MATH Level 5": 0.0174, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4245, + "hfopenllm_v2/MMLU-PRO": 0.4207 } }, { @@ -67456,12 +67476,12 @@ "developer": "rombodawg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2566, - "hfopenllm_v2/BBH": 0.39, - "hfopenllm_v2/MATH Level 5": 0.1208, - "hfopenllm_v2/GPQA": 0.2626, + "hfopenllm_v2/IFEval": 0.2595, + "hfopenllm_v2/BBH": 0.3884, + "hfopenllm_v2/MATH Level 5": 0.0914, + "hfopenllm_v2/GPQA": 0.2743, "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.2741 + "hfopenllm_v2/MMLU-PRO": 0.2719 } }, { @@ -69573,12 +69593,12 @@ "developer": "spow12", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6511, - "hfopenllm_v2/BBH": 0.5926, - "hfopenllm_v2/MATH Level 5": 0.1858, - "hfopenllm_v2/GPQA": 0.3247, + "hfopenllm_v2/IFEval": 0.6517, + "hfopenllm_v2/BBH": 0.5908, + "hfopenllm_v2/MATH Level 5": 0.2032, + "hfopenllm_v2/GPQA": 0.3238, "hfopenllm_v2/MUSR": 0.3842, - "hfopenllm_v2/MMLU-PRO": 0.3836 + "hfopenllm_v2/MMLU-PRO": 0.3812 } }, { @@ -72473,6 +72493,15 @@ "hfopenllm_v2/MMLU-PRO": 0.2992 } }, + { + "id": "utter-project/EuroLLM-9B", + "name": "EuroLLM 9B", + "developer": "unknown", + "evaluator_relationship": null, + "benchmark_scores": { + "la_leaderboard/la_leaderboard": 25.87 + } + }, { "id": "uukuguy/speechless-code-mistral-7b-v1.0", "name": "speechless-code-mistral-7b-v1.0", @@ -73231,17 +73260,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7982, + "reward-bench/Score": 0.596, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Safety": 0.6911, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753 + "reward-bench/Ties": 0.6226 } }, { @@ -73733,7 +73762,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 23.1 + "terminal-bench-2.0/terminal-bench-2.0": 25.4 } }, { @@ -73775,7 +73804,7 @@ "developer": "xAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 14.2 + "terminal-bench-2.0/terminal-bench-2.0": 25.8 } }, { @@ -74134,12 +74163,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } }, { diff --git a/data/models/AtAndDev_Qwen2.5-1.5B-continuous-learnt.json b/data/models/AtAndDev_Qwen2.5-1.5B-continuous-learnt.json index dba26133fd6dfa178c728b102613f0a5db478958..8aa5dd3c1d9edc9b6735b92c14cc680d8f5aedfa 100644 --- a/data/models/AtAndDev_Qwen2.5-1.5B-continuous-learnt.json +++ b/data/models/AtAndDev_Qwen2.5-1.5B-continuous-learnt.json @@ -5,7 +5,7 @@ "developer": "AtAndDev", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "1.544" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4605 + "score": 0.4511 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4258 + "score": 0.4275 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0748 + "score": 0.1473 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3636 + "score": 0.3623 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.2806 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4511 + "score": 0.4605 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4275 + "score": 0.4258 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1473 + "score": 0.0748 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3623 + "score": 0.3636 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2806 + "score": 0.2812 } } ], diff --git a/data/models/CIR-AMS_BTRM_Qwen2_7b_0613.json b/data/models/CIR-AMS_BTRM_Qwen2_7b_0613.json index 9fb827ce32fb32044e2247d7f86c70d1bc13d414..84a36ab2262aa5d4869e5d141d383b85699971f1 100644 --- a/data/models/CIR-AMS_BTRM_Qwen2_7b_0613.json +++ b/data/models/CIR-AMS_BTRM_Qwen2_7b_0613.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5736 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.8172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3563 + "score": 0.9749 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7178 + "score": 0.9014 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5737 + "score": 0.8775 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6527 + "score": 0.7029 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8172 + "score": 0.5736 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9749 + "score": 0.5347 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3563 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6066 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9014 + "score": 0.7178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8775 + "score": 0.5737 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7029 + "score": 0.6527 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/Daemontatox_AetherTOT.json b/data/models/Daemontatox_AetherTOT.json index 2ea373c2eebb0dfbd13abbe17934735c92168866..72bdb01863fd9cb8a56697def402922f2204f3a5 100644 --- a/data/models/Daemontatox_AetherTOT.json +++ b/data/models/Daemontatox_AetherTOT.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MllamaForConditionalGeneration", "params_billions": "10.67" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4398 + "score": 0.4383 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5066 + "score": 0.5034 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1488 + "score": 0.1443 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4079 + "score": 0.4052 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3804 + "score": 0.3778 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4383 + "score": 0.4398 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5034 + "score": 0.5066 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1443 + "score": 0.1488 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4052 + "score": 0.4079 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3778 + "score": 0.3804 } } ], diff --git a/data/models/Daemontatox_PathfinderAI.json b/data/models/Daemontatox_PathfinderAI.json index cea390cdba1611d28ac0ecfbae5b1930078d8645..e1f5c19d36675adf14e4da07a19941ae0f69be33 100644 --- a/data/models/Daemontatox_PathfinderAI.json +++ b/data/models/Daemontatox_PathfinderAI.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4855 + "score": 0.3745 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6627 + "score": 0.6668 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4841 + "score": 0.4758 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3943 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4256 + "score": 0.4858 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5542 + "score": 0.5593 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3745 + "score": 0.4855 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6668 + "score": 0.6627 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4841 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.3096 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4858 + "score": 0.4256 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5593 + "score": 0.5542 } } ], diff --git a/data/models/DavieLion_Llama-3.2-1B-SPIN-iter0.json b/data/models/DavieLion_Llama-3.2-1B-SPIN-iter0.json index de774747911df630678aaeeb62f20bc722910dbb..34acafe2182ed5fe43de3f09e1727813e659e101 100644 --- a/data/models/DavieLion_Llama-3.2-1B-SPIN-iter0.json +++ b/data/models/DavieLion_Llama-3.2-1B-SPIN-iter0.json @@ -5,7 +5,7 @@ "developer": "DavieLion", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1549 + "score": 0.1507 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2937 + "score": 0.293 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.006 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2534 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1125 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1507 + "score": 0.1549 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.293 + "score": 0.2937 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.006 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2576 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1125 + "score": 0.1128 } } ], diff --git a/data/models/Goekdeniz-Guelmez_Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1.json b/data/models/Goekdeniz-Guelmez_Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1.json index 920f7b164dd643d5ff9c56a4a81b33afd9ac2f5d..e452475d2b1aadeb10e17685131a9a96db77f7b2 100644 --- a/data/models/Goekdeniz-Guelmez_Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1.json +++ b/data/models/Goekdeniz-Guelmez_Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "0.63" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3417 + "score": 0.3472 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3268 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0023 + "score": 0.0891 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2517 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3249 + "score": 0.3262 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1638 + "score": 0.1641 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3472 + "score": 0.3417 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3268 + "score": 0.3292 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0891 + "score": 0.0023 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2576 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3262 + "score": 0.3249 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1641 + "score": 0.1638 } } ], diff --git a/data/models/Goekdeniz-Guelmez_josie-7b-v6.0-step2000.json b/data/models/Goekdeniz-Guelmez_josie-7b-v6.0-step2000.json index 3392ce460e2a7edf26ac0b55d3f6ec69383c525b..51ca3ee7e1e11069a115661541b7f8295515c3c6 100644 --- a/data/models/Goekdeniz-Guelmez_josie-7b-v6.0-step2000.json +++ b/data/models/Goekdeniz-Guelmez_josie-7b-v6.0-step2000.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7628 + "score": 0.7598 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5107 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.4237 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2802 + "score": 0.2768 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4579 + "score": 0.4539 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4033 + "score": 0.4012 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7598 + "score": 0.7628 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5107 + "score": 0.5098 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4237 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2802 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4539 + "score": 0.4579 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4012 + "score": 0.4033 } } ], diff --git a/data/models/Gunulhona_Gemma-Ko-Merge-PEFT.json b/data/models/Gunulhona_Gemma-Ko-Merge-PEFT.json index 471cbc40f26f25880bbbadc1c0b58c7bb92fd5a0..4076976ed7893aae438b45c4ef00cd9c900634a8 100644 --- a/data/models/Gunulhona_Gemma-Ko-Merge-PEFT.json +++ b/data/models/Gunulhona_Gemma-Ko-Merge-PEFT.json @@ -5,7 +5,7 @@ "developer": "Gunulhona", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "?", "params_billions": "20.318" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4441 + "score": 0.288 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4863 + "score": 0.5154 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.307 + "score": 0.3247 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3986 + "score": 0.408 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3098 + "score": 0.3817 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.288 + "score": 0.4441 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5154 + "score": 0.4863 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.307 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.408 + "score": 0.3986 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3817 + "score": 0.3098 } } ], diff --git a/data/models/HuggingFaceTB_SmolLM2-135M-Instruct.json b/data/models/HuggingFaceTB_SmolLM2-135M-Instruct.json index 455442fa21083c2481d086a20672b2502b54f591..a638958edde39c7abd8f5124d7ebfa00c20c75ea 100644 --- a/data/models/HuggingFaceTB_SmolLM2-135M-Instruct.json +++ b/data/models/HuggingFaceTB_SmolLM2-135M-Instruct.json @@ -5,7 +5,7 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "0.135" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2883 + "score": 0.0593 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3124 + "score": 0.3135 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.003 + "score": 0.0144 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2357 + "score": 0.2341 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3662 + "score": 0.3871 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1115 + "score": 0.1092 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0593 + "score": 0.2883 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3135 + "score": 0.3124 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0144 + "score": 0.003 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2341 + "score": 0.2357 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3871 + "score": 0.3662 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1092 + "score": 0.1115 } } ], diff --git a/data/models/HuggingFaceTB_SmolLM2-360M-Instruct.json b/data/models/HuggingFaceTB_SmolLM2-360M-Instruct.json index 306a3120d2b6ac00c169dc4bdb94b3d021238634..090f11323fa59955f3ec139c4205742612722b97 100644 --- a/data/models/HuggingFaceTB_SmolLM2-360M-Instruct.json +++ b/data/models/HuggingFaceTB_SmolLM2-360M-Instruct.json @@ -5,9 +5,9 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "0.362" + "params_billions": "0.36" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.083 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3144 + "score": 0.3053 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0151 + "score": 0.0083 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.255 + "score": 0.2651 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3461 + "score": 0.3423 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1117 + "score": 0.1126 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.083 + "score": 0.3842 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3053 + "score": 0.3144 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0083 + "score": 0.0151 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2651 + "score": 0.255 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3423 + "score": 0.3461 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1117 } } ], diff --git a/data/models/Isaak-Carter_JOSIEv4o-8b-stage1-v4.json b/data/models/Isaak-Carter_JOSIEv4o-8b-stage1-v4.json index aa9474affd3265670ded4b39e44c5307e090be44..7b49905b129cd681075f7d22b1d70d075f45f067 100644 --- a/data/models/Isaak-Carter_JOSIEv4o-8b-stage1-v4.json +++ b/data/models/Isaak-Carter_JOSIEv4o-8b-stage1-v4.json @@ -5,7 +5,7 @@ "developer": "Isaak-Carter", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2477 + "score": 0.2553 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4725 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0453 + "score": 0.0529 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2911 + "score": 0.2919 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3641 + "score": 0.3654 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3316 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2553 + "score": 0.2477 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4725 + "score": 0.4758 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0529 + "score": 0.0453 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2919 + "score": 0.2911 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3654 + "score": 0.3641 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3316 + "score": 0.3292 } } ], diff --git a/data/models/LxzGordon_URM-LLaMa-3.1-8B.json b/data/models/LxzGordon_URM-LLaMa-3.1-8B.json index 7f03035f2bd809fb14be27130e313f5b86a26f9a..2ce56c90ce0d5fb5788660f5b3f1f1e179701786 100644 --- a/data/models/LxzGordon_URM-LLaMa-3.1-8B.json +++ b/data/models/LxzGordon_URM-LLaMa-3.1-8B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7394 + "score": 0.9294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6884 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.45 + "score": 0.8816 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6393 + "score": 0.9108 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9178 + "score": 0.9698 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9758 + "score": 0.7394 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7653 + "score": 0.6884 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9294 + "score": 0.45 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.6393 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8816 + "score": 0.9178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9108 + "score": 0.9758 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9698 + "score": 0.7653 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/Magpie-Align_Llama-3-8B-Magpie-Align-v0.1.json b/data/models/Magpie-Align_Llama-3-8B-Magpie-Align-v0.1.json index 4ff754d779a37f976dff5ceb14d8ae84536fe5e6..345c65d31a7dd5eb78e83a760ceabf92a149d34c 100644 --- a/data/models/Magpie-Align_Llama-3-8B-Magpie-Align-v0.1.json +++ b/data/models/Magpie-Align_Llama-3-8B-Magpie-Align-v0.1.json @@ -5,7 +5,7 @@ "developer": "Magpie-Align", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4027 + "score": 0.4118 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.4811 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0461 + "score": 0.034 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2752 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3047 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3001 + "score": 0.3006 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4118 + "score": 0.4027 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4811 + "score": 0.4789 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.034 + "score": 0.0461 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.2768 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3047 + "score": 0.3087 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3006 + "score": 0.3001 } } ], diff --git a/data/models/NCSOFT_Llama-3-OffsetBias-RM-8B.json b/data/models/NCSOFT_Llama-3-OffsetBias-RM-8B.json index 0a839fbe35d3881341976e6cd6ecdba200fd0827..14fabae93f5ca8f04e2a0c2d895536cd5cdfcb3a 100644 --- a/data/models/NCSOFT_Llama-3-OffsetBias-RM-8B.json +++ b/data/models/NCSOFT_Llama-3-OffsetBias-RM-8B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", + "evaluation_id": "reward-bench-2/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8942 + "score": 0.648 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9721 + "score": 0.6084 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.818 + "score": 0.4 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9192 + "score": 0.7222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.648 + "score": 0.9596 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6084 + "score": 0.6786 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/NCSOFT_Llama-3-OffsetBias-RM-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4 + "score": 0.8942 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.9721 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7222 + "score": 0.818 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9596 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6786 + "score": 0.9192 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/Nexusflow_Starling-RM-34B.json b/data/models/Nexusflow_Starling-RM-34B.json index 8ab4f392fbf3047e7bcb88c6ed9f3a2b6d8e5a37..0373bd963cc9795a2ad38fc9da30f8417d64ff6d 100644 --- a/data/models/Nexusflow_Starling-RM-34B.json +++ b/data/models/Nexusflow_Starling-RM-34B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4553 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4589 + "score": 0.8133 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3187 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6175 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7556 + "score": 0.877 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4808 + "score": 0.8845 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1004 + "score": 0.7137 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Nexusflow_Starling-RM-34B/1766412838.146816", + "evaluation_id": "reward-bench-2/Nexusflow_Starling-RM-34B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8133 + "score": 0.4553 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.4589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3187 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6175 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.877 + "score": 0.7556 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8845 + "score": 0.4808 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7137 + "score": 0.1004 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/Omkar1102_code-yi.json b/data/models/Omkar1102_code-yi.json index 218808f7891ec6cca4849b1bac865cbc2cdea8ef..006e593635727bc527240326fdfe365ce05e0044 100644 --- a/data/models/Omkar1102_code-yi.json +++ b/data/models/Omkar1102_code-yi.json @@ -5,7 +5,7 @@ "developer": "Omkar1102", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "2.084" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2148 + "score": 0.2254 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.275 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2508 + "score": 0.2576 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3762 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1123 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2254 + "score": 0.2148 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.276 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2508 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3762 + "score": 0.3802 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1123 + "score": 0.1126 } } ], diff --git a/data/models/OpenAssistant_reward-model-deberta-v3-large-v2.json b/data/models/OpenAssistant_reward-model-deberta-v3-large-v2.json index b28ca5c5700af0f5cf22d77dcb1c4fea033ecc2d..cf1ba02f6dd573bf3e0770ff614660b73e92dbb8 100644 --- a/data/models/OpenAssistant_reward-model-deberta-v3-large-v2.json +++ b/data/models/OpenAssistant_reward-model-deberta-v3-large-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6126 + "score": 0.32 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8939 + "score": 0.3853 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4518 + "score": 0.2687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5027 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7338 + "score": 0.3667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3855 + "score": 0.2768 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5836 + "score": 0.12 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_reward-model-deberta-v3-large-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.32 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3853 + "score": 0.6126 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2687 + "score": 0.8939 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.4518 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3667 + "score": 0.7338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.3855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.12 + "score": 0.5836 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/PKU-Alignment_beaver-7b-v1.0-cost.json b/data/models/PKU-Alignment_beaver-7b-v1.0-cost.json index 8e786484059e3101c1249c1cce8c4b2be81faa5a..3777eba3edfdc470c669a503ac85994bf8139135 100644 --- a/data/models/PKU-Alignment_beaver-7b-v1.0-cost.json +++ b/data/models/PKU-Alignment_beaver-7b-v1.0-cost.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5798 + "score": 0.3332 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6173 + "score": 0.3263 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.2313 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3989 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7351 + "score": 0.7589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5482 + "score": 0.2939 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3332 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.5798 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2313 + "score": 0.6173 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3989 + "score": 0.4232 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7589 + "score": 0.7351 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2939 + "score": 0.5482 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.57 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/PKU-Alignment_beaver-7b-v1.0-reward.json b/data/models/PKU-Alignment_beaver-7b-v1.0-reward.json index adf890d6a20deeb5311651870c4e5638866b9733..ee66fd95461643d78856676dc7b45eb953fc109f 100644 --- a/data/models/PKU-Alignment_beaver-7b-v1.0-reward.json +++ b/data/models/PKU-Alignment_beaver-7b-v1.0-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.1606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2105 + "score": 0.4727 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2938 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2623 + "score": 0.2873 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1422 + "score": 0.3757 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0646 + "score": 0.346 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.5993 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4727 + "score": 0.1606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.2105 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2873 + "score": 0.2938 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.2623 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3757 + "score": 0.1422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.0646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5993 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/PKU-Alignment_beaver-7b-v2.0-cost.json b/data/models/PKU-Alignment_beaver-7b-v2.0-cost.json index 81393cc7dd76e676c977d4915c4171a2a2ea4f5e..5314c675193b80f86e094f04e503749d966c642b 100644 --- a/data/models/PKU-Alignment_beaver-7b-v2.0-cost.json +++ b/data/models/PKU-Alignment_beaver-7b-v2.0-cost.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v2.0-cost/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v2.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5957 + "score": 0.3326 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5726 + "score": 0.3789 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4561 + "score": 0.275 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3333 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7608 + "score": 0.7356 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6211 + "score": 0.2828 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5397 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v2.0-cost/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v2.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3326 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3789 + "score": 0.5957 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.5726 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3333 + "score": 0.4561 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7356 + "score": 0.7608 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2828 + "score": 0.6211 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.5397 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/Quazim0t0_Casa-14b-sce.json b/data/models/Quazim0t0_Casa-14b-sce.json index 4925641638cc022bfd4ce69c07ae8478822ca4ab..b9c8b5212e5669beb41b66ce1f966b05f5958e6c 100644 --- a/data/models/Quazim0t0_Casa-14b-sce.json +++ b/data/models/Quazim0t0_Casa-14b-sce.json @@ -5,7 +5,7 @@ "developer": "Quazim0t0", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "14.66" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6654 + "score": 0.6718 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6901 + "score": 0.6891 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4698 + "score": 0.4985 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3331 + "score": 0.3339 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.431 + "score": 0.4323 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5426 + "score": 0.5408 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6718 + "score": 0.6654 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6891 + "score": 0.6901 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4985 + "score": 0.4698 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3339 + "score": 0.3331 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4323 + "score": 0.431 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5408 + "score": 0.5426 } } ], diff --git a/data/models/Qwen_Qwen2.5-0.5B-Instruct.json b/data/models/Qwen_Qwen2.5-0.5B-Instruct.json index 3d0aedd0be81cb65537fa8994b8b471d9f8ba0cf..ff626bd43f5a0267c85cd8e28dde66f183db6457 100644 --- a/data/models/Qwen_Qwen2.5-0.5B-Instruct.json +++ b/data/models/Qwen_Qwen2.5-0.5B-Instruct.json @@ -5,9 +5,9 @@ "developer": "Qwen", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", - "params_billions": "0.5" + "params_billions": "0.494" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3153 + "score": 0.3071 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3322 + "score": 0.3341 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1035 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2576 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3342 + "score": 0.3329 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.172 + "score": 0.1697 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3071 + "score": 0.3153 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3341 + "score": 0.3322 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1035 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2592 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3329 + "score": 0.3342 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1697 + "score": 0.172 } } ], diff --git a/data/models/Qwen_Qwen2.5-7B.json b/data/models/Qwen_Qwen2.5-7B.json index e60b70e29b23d4b7de9d8ba6b77d947e2643e4d4..6d34d0ed07ce4e28a6c008cc60b3e1c8a872068c 100644 --- a/data/models/Qwen_Qwen2.5-7B.json +++ b/data/models/Qwen_Qwen2.5-7B.json @@ -1,14 +1,8 @@ { "model_info": { - "name": "Qwen2.5-7B", + "name": "Qwen2.5 7B", "id": "Qwen/Qwen2.5-7B", - "developer": "Qwen", - "inference_platform": "unknown", - "additional_details": { - "precision": "bfloat16", - "architecture": "Qwen2ForCausalLM", - "params_billions": "7.616" - } + "developer": "unknown" }, "evaluations": [ { @@ -140,6 +134,46 @@ ], "detailed_evaluation_results": null, "generation_config": null + }, + { + "evaluation_id": "la_leaderboard/Qwen/Qwen2.5-7B/1774451270", + "retrieved_timestamp": "2024-10-27T00:00:00Z", + "source_metadata": { + "source_name": "La Leaderboard", + "source_type": "evaluation_run", + "source_url": "https://huggingface.co/spaces/la-leaderboard/la-leaderboard", + "source_organization_name": "La Leaderboard", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "custom", + "version": "1.0" + }, + "benchmark": "la_leaderboard", + "evaluation_results": [ + { + "evaluation_name": "la_leaderboard", + "metric_config": { + "evaluation_description": "La Leaderboard: LLM evaluation for Spanish varieties and languages of Spain and Latin America", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 100 + }, + "score_details": { + "score": 27.61 + }, + "source_data": { + "source_type": "url", + "dataset_name": "La Leaderboard composite dataset", + "url": [ + "https://huggingface.co/spaces/la-leaderboard/la-leaderboard" + ] + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null } ] } \ No newline at end of file diff --git a/data/models/Qwen_Qwen2.5-Coder-7B-Instruct.json b/data/models/Qwen_Qwen2.5-Coder-7B-Instruct.json index 314a801fd64eb46e8adee5d8adcc5e0a741856c0..ea1b32a72ef0c31b1f58a3f65a1291ede3e861c2 100644 --- a/data/models/Qwen_Qwen2.5-Coder-7B-Instruct.json +++ b/data/models/Qwen_Qwen2.5-Coder-7B-Instruct.json @@ -5,7 +5,7 @@ "developer": "Qwen", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6147 + "score": 0.6101 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4999 + "score": 0.5008 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.3716 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2936 + "score": 0.2919 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.4073 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.3352 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6101 + "score": 0.6147 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5008 + "score": 0.4999 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.031 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2919 + "score": 0.2936 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4073 + "score": 0.4099 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3352 + "score": 0.3354 } } ], diff --git a/data/models/Ray2333_GRM-gemma2-2B-rewardmodel-ft.json b/data/models/Ray2333_GRM-gemma2-2B-rewardmodel-ft.json index 24dd55ac376e51965493e81885df295559db7184..cd8835eaf989dbd622c51262fef9b72b10534d67 100644 --- a/data/models/Ray2333_GRM-gemma2-2B-rewardmodel-ft.json +++ b/data/models/Ray2333_GRM-gemma2-2B-rewardmodel-ft.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5966 + "score": 0.8839 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5305 + "score": 0.9302 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3125 + "score": 0.7719 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5902 + "score": 0.9216 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.912 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7455 + "score": 0.5966 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4788 + "score": 0.5305 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8839 + "score": 0.3125 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.5902 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7719 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9216 + "score": 0.7455 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.912 + "score": 0.4788 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/Replete-AI_Replete-LLM-Qwen2-7b.json b/data/models/Replete-AI_Replete-LLM-Qwen2-7b.json index 4c1a2520189dd469c6d0a6c10b5887e7aab7aadc..df8c220914f8ca0c56ba74871850b48949972c43 100644 --- a/data/models/Replete-AI_Replete-LLM-Qwen2-7b.json +++ b/data/models/Replete-AI_Replete-LLM-Qwen2-7b.json @@ -5,7 +5,7 @@ "developer": "Replete-AI", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0932 + "score": 0.0905 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2985 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2475 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3941 + "score": 0.3848 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1157 + "score": 0.1158 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0905 + "score": 0.0932 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2985 + "score": 0.2977 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2475 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3848 + "score": 0.3941 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1158 + "score": 0.1157 } } ], diff --git a/data/models/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1.json b/data/models/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1.json index 988f599afd76311b87e49f16160d0ca534436cdd..4a85b178092e7e5acaf395eb054cb8adc61e2891 100644 --- a/data/models/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1.json +++ b/data/models/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7249 + "score": 0.9499 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7558 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6448 + "score": 0.9378 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.9903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9131 + "score": 0.7249 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7633 + "score": 0.7558 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9499 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.6448 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9079 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9378 + "score": 0.9131 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9903 + "score": 0.7633 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/Skywork_Skywork-Reward-Gemma-2-27B.json b/data/models/Skywork_Skywork-Reward-Gemma-2-27B.json index c2b742dbc17aa9720f66d090e08eb784ec7accdc..0b4cc11a7f17a4535cc37f935fdae034a6214bce 100644 --- a/data/models/Skywork_Skywork-Reward-Gemma-2-27B.json +++ b/data/models/Skywork_Skywork-Reward-Gemma-2-27B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7576 + "score": 0.938 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7368 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4031 + "score": 0.9145 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7049 + "score": 0.9189 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9422 + "score": 0.9606 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9323 + "score": 0.7576 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8261 + "score": 0.7368 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.938 + "score": 0.4031 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.7049 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9145 + "score": 0.9422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9189 + "score": 0.9323 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9606 + "score": 0.8261 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/UCLA-AGI_Llama-3-Instruct-8B-SPPO-Iter3.json b/data/models/UCLA-AGI_Llama-3-Instruct-8B-SPPO-Iter3.json index 2226a318a85cc821aef97dd758e350fdd8de0974..5aa93e3f9ffc82a9b47329510f56c5672e33e6e6 100644 --- a/data/models/UCLA-AGI_Llama-3-Instruct-8B-SPPO-Iter3.json +++ b/data/models/UCLA-AGI_Llama-3-Instruct-8B-SPPO-Iter3.json @@ -5,7 +5,7 @@ "developer": "UCLA-AGI", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6703 + "score": 0.6834 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5076 + "score": 0.508 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0718 + "score": 0.0959 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3647 + "score": 0.3661 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3658 + "score": 0.3644 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6834 + "score": 0.6703 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.508 + "score": 0.5076 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0959 + "score": 0.0718 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3661 + "score": 0.3647 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3644 + "score": 0.3658 } } ], diff --git a/data/models/ValiantLabs_Llama3.1-8B-ShiningValiant2.json b/data/models/ValiantLabs_Llama3.1-8B-ShiningValiant2.json index b9a3d8795985f8b725e75e899a6dc5b39573783a..c82e658291ec493a85f6ef296178fd95b5dd7ec3 100644 --- a/data/models/ValiantLabs_Llama3.1-8B-ShiningValiant2.json +++ b/data/models/ValiantLabs_Llama3.1-8B-ShiningValiant2.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2678 + "score": 0.6496 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4429 + "score": 0.4774 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0566 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.302 + "score": 0.3104 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3909 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2927 + "score": 0.3382 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6496 + "score": 0.2678 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4774 + "score": 0.4429 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0566 + "score": 0.0521 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.302 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3909 + "score": 0.3959 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3382 + "score": 0.2927 } } ], diff --git a/data/models/abhishek_autotrain-0tmgq-5tpbg.json b/data/models/abhishek_autotrain-0tmgq-5tpbg.json index 0e0ab50b4d7d638e4b16e830f0b233023f376915..5b7e33c34389c98f5f526e52c2328ef7923902ba 100644 --- a/data/models/abhishek_autotrain-0tmgq-5tpbg.json +++ b/data/models/abhishek_autotrain-0tmgq-5tpbg.json @@ -5,7 +5,7 @@ "developer": "abhishek", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "0.135" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1957 + "score": 0.1952 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3135 + "score": 0.3127 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0128 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2592 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.365 + "score": 0.3584 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1151 + "score": 0.1144 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1952 + "score": 0.1957 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3127 + "score": 0.3135 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0128 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2517 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3584 + "score": 0.365 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1144 + "score": 0.1151 } } ], diff --git a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json index a96a6a2c58b30640ee0de43b8043a05f02c8d033..b0aa28b87dd09302e2563580f7e86b9555a6d01e 100644 --- a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json +++ b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json @@ -38,7 +38,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7058 + "score": 0.6905 }, "source_data": { "dataset_name": "RewardBench", @@ -56,7 +56,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9525 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -74,7 +74,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3947 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -92,7 +92,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7703 + "score": 0.7676 }, "source_data": { "dataset_name": "RewardBench", @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7004 + "score": 0.7058 }, "source_data": { "dataset_name": "RewardBench", @@ -152,7 +152,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9413 + "score": 0.9525 }, "source_data": { "dataset_name": "RewardBench", @@ -170,7 +170,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3947 }, "source_data": { "dataset_name": "RewardBench", @@ -188,7 +188,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7716 + "score": 0.7703 }, "source_data": { "dataset_name": "RewardBench", @@ -230,7 +230,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6945 + "score": 0.7004 }, "source_data": { "dataset_name": "RewardBench", @@ -248,7 +248,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9413 }, "source_data": { "dataset_name": "RewardBench", @@ -266,7 +266,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -284,7 +284,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7743 + "score": 0.7716 }, "source_data": { "dataset_name": "RewardBench", @@ -326,7 +326,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6924 + "score": 0.6808 }, "source_data": { "dataset_name": "RewardBench", @@ -344,7 +344,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9302 }, "source_data": { "dataset_name": "RewardBench", @@ -362,7 +362,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3575 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -380,7 +380,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7527 }, "source_data": { "dataset_name": "RewardBench", @@ -422,7 +422,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7008 + "score": 0.6945 }, "source_data": { "dataset_name": "RewardBench", @@ -458,7 +458,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -476,7 +476,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7743 }, "source_data": { "dataset_name": "RewardBench", @@ -518,7 +518,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7019 + "score": 0.6895 }, "source_data": { "dataset_name": "RewardBench", @@ -536,7 +536,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -554,7 +554,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.375 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -572,7 +572,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7811 + "score": 0.7595 }, "source_data": { "dataset_name": "RewardBench", @@ -614,7 +614,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6808 + "score": 0.7019 }, "source_data": { "dataset_name": "RewardBench", @@ -632,7 +632,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.9497 }, "source_data": { "dataset_name": "RewardBench", @@ -650,7 +650,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.375 }, "source_data": { "dataset_name": "RewardBench", @@ -668,7 +668,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7527 + "score": 0.7811 }, "source_data": { "dataset_name": "RewardBench", @@ -710,7 +710,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6905 + "score": 0.6924 }, "source_data": { "dataset_name": "RewardBench", @@ -746,7 +746,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3575 }, "source_data": { "dataset_name": "RewardBench", @@ -764,7 +764,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7676 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -806,7 +806,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6895 + "score": 0.7008 }, "source_data": { "dataset_name": "RewardBench", @@ -842,7 +842,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -860,7 +860,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7595 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", diff --git a/data/models/alibaba_qwen-3-coder-480b.json b/data/models/alibaba_qwen-3-coder-480b.json index f950751e93e00355bc7f7e7d39efd2dbbec7d40a..af103b1b43bc3a288d48697942f3b3a30a04080b 100644 --- a/data/models/alibaba_qwen-3-coder-480b.json +++ b/data/models/alibaba_qwen-3-coder-480b.json @@ -4,8 +4,8 @@ "id": "alibaba/qwen-3-coder-480b", "developer": "Alibaba", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Dakou Agent", + "agent_organization": "iflow" } }, "evaluations": [ @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-28", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-12-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 23.9, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json b/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json index 73411ab4fcdf3bfd9a55ece3006bf091682be221..18e9257ee4a295645c5d70b31ea729844e529c44 100644 --- a/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json +++ b/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/allenai_Llama-3.1-70B-Instruct-RM-RB2.json b/data/models/allenai_Llama-3.1-70B-Instruct-RM-RB2.json index 6b82bed55df1bdb62067f41e1f2f588c7fd5e772..3edb4e861c9c885f1d8b857f90568f09968d9fbc 100644 --- a/data/models/allenai_Llama-3.1-70B-Instruct-RM-RB2.json +++ b/data/models/allenai_Llama-3.1-70B-Instruct-RM-RB2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.7606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8126 + "score": 0.9021 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.8355 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8844 + "score": 0.9095 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8646 + "score": 0.8969 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8835 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-70B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9021 + "score": 0.7606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.8126 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8355 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6995 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9095 + "score": 0.8844 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8969 + "score": 0.8646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8835 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_Llama-3.1-8B-Base-RM-RB2.json b/data/models/allenai_Llama-3.1-8B-Base-RM-RB2.json index b77566d88bfa5b7f18f1a64562aee1f65b27a08c..8e81065be5c6f388ddbf031b25c6ea031d346e21 100644 --- a/data/models/allenai_Llama-3.1-8B-Base-RM-RB2.json +++ b/data/models/allenai_Llama-3.1-8B-Base-RM-RB2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8463 + "score": 0.649 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.933 + "score": 0.72 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7785 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.612 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8851 + "score": 0.8267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7886 + "score": 0.8323 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.5406 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.649 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.72 + "score": 0.8463 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.933 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.612 + "score": 0.7785 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8851 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8323 + "score": 0.7886 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/allenai_Llama-3.1-8B-Instruct-RM-RB2.json b/data/models/allenai_Llama-3.1-8B-Instruct-RM-RB2.json index 60d204122c692e04be6211b26f2d79d76e83f431..2b424dce10a93f404c9945de99ddd768647b6184 100644 --- a/data/models/allenai_Llama-3.1-8B-Instruct-RM-RB2.json +++ b/data/models/allenai_Llama-3.1-8B-Instruct-RM-RB2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.7285 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7432 + "score": 0.8885 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4437 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6175 + "score": 0.8158 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8956 + "score": 0.8932 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9071 + "score": 0.887 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7638 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Instruct-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Instruct-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8885 + "score": 0.7285 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.7432 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8158 + "score": 0.4437 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6175 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8932 + "score": 0.8956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.887 + "score": 0.9071 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.7638 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2.json b/data/models/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2.json index 56c7eedcfadd2447274047e1824ef99e14cf4e28..3c470e6abd70f5f2bc9a8be57cc36f939ca7753c 100644 --- a/data/models/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2.json +++ b/data/models/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.722 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8084 + "score": 0.8892 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3688 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.8268 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8689 + "score": 0.9027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7778 + "score": 0.8583 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8308 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8892 + "score": 0.722 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.8084 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8268 + "score": 0.3688 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6776 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9027 + "score": 0.8689 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8583 + "score": 0.7778 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8308 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_Llama-3.1-Tulu-3-70B.json b/data/models/allenai_Llama-3.1-Tulu-3-70B.json index 37cdffe627a7cf3591534152856ebba483540143..fb86cb10e4b1586a637369168cb0a7759a2a884f 100644 --- a/data/models/allenai_Llama-3.1-Tulu-3-70B.json +++ b/data/models/allenai_Llama-3.1-Tulu-3-70B.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "70.554" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8379 + "score": 0.8291 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6157 + "score": 0.6164 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3829 + "score": 0.4502 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4988 + "score": 0.4948 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4656 + "score": 0.4645 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8291 + "score": 0.8379 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6164 + "score": 0.6157 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4502 + "score": 0.3829 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4948 + "score": 0.4988 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4645 + "score": 0.4656 } } ], diff --git a/data/models/allenai_Llama-3.1-Tulu-3-8B.json b/data/models/allenai_Llama-3.1-Tulu-3-8B.json index cdc4277bf3338da5223bb18b774ff35b0fb5f41e..3ad8a9ac23bd0436da7fc124729f9555d65628da 100644 --- a/data/models/allenai_Llama-3.1-Tulu-3-8B.json +++ b/data/models/allenai_Llama-3.1-Tulu-3-8B.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8255 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.405 + "score": 0.4061 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1964 + "score": 0.2115 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2987 + "score": 0.297 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2821 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8255 + "score": 0.8267 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4061 + "score": 0.405 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2115 + "score": 0.1964 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.297 + "score": 0.2987 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2821 + "score": 0.2827 } } ], diff --git a/data/models/anthropic_Opus_4.5.json b/data/models/anthropic_Opus_4.5.json index 7588f6974f00aa572cbc95b6810e211a97ba6f0d..c02985e498ba1b07a18f09640000b046397f1520 100644 --- a/data/models/anthropic_Opus_4.5.json +++ b/data/models/anthropic_Opus_4.5.json @@ -6,76 +6,6 @@ "inference_platform": "unknown" }, "evaluations": [ - { - "evaluation_id": "ace/anthropic_opus-4.5/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Overall Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Overall ACE score (paper snapshot).", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.478 - }, - "generation_config": { - "additional_details": { - "run_setting": "On" - } - } - }, - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.391 - }, - "generation_config": { - "additional_details": { - "run_setting": "On" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "On" - } - } - }, { "evaluation_id": "apex-agents/anthropic_opus-4.5/1773260200", "retrieved_timestamp": "1773260200", @@ -275,6 +205,76 @@ } } }, + { + "evaluation_id": "ace/anthropic_opus-4.5/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Overall Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Overall ACE score (paper snapshot).", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.478 + }, + "generation_config": { + "additional_details": { + "run_setting": "On" + } + } + }, + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.391 + }, + "generation_config": { + "additional_details": { + "run_setting": "On" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "On" + } + } + }, { "evaluation_id": "apex-v1/anthropic_opus-4.5/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/anthropic_claude-3-5-haiku-20241022.json b/data/models/anthropic_claude-3-5-haiku-20241022.json index d77bc43873534d2017192f9adf45a50bd095f489..40d89ab29be589f81577a4b7da1462f24631eedf 100644 --- a/data/models/anthropic_claude-3-5-haiku-20241022.json +++ b/data/models/anthropic_claude-3-5-haiku-20241022.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-7-sonnet-20250219.json b/data/models/anthropic_claude-3-7-sonnet-20250219.json index 90fc471d7596ee282f6a9f14697b4fb36cae70f7..c65b70d499889538df9925f7bb281f5fd976eb6c 100644 --- a/data/models/anthropic_claude-3-7-sonnet-20250219.json +++ b/data/models/anthropic_claude-3-7-sonnet-20250219.json @@ -9,8 +9,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -524,8 +524,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-haiku-4.5.json b/data/models/anthropic_claude-haiku-4.5.json index e264d494135c2a9e41ab0ec25ddb1450437254d8..def65baa0a3f8cecb7548c5dc2227e66fdd3c699 100644 --- a/data/models/anthropic_claude-haiku-4.5.json +++ b/data/models/anthropic_claude-haiku-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-haiku-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Goose", - "agent_organization": "Block" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.8, + "score": 28.3, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.5, + "score": 35.5, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 28.3, + "score": 27.5, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.5, + "score": 29.8, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4-1-20250805.json b/data/models/anthropic_claude-opus-4-1-20250805.json index 07cff8b0011fcb0f61f1699481df5c21619b9081..476b4d0fe98c0347ea36fcd24c9b417b2249c4d2 100644 --- a/data/models/anthropic_claude-opus-4-1-20250805.json +++ b/data/models/anthropic_claude-opus-4-1-20250805.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-opus-4-1-20250805/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-opus-4-5.json b/data/models/anthropic_claude-opus-4-5.json index 4e4da6a5a225e683bd80d8e74b30768f496388d6..b814382573e3978646fd16b279ebd76181a6d66a 100644 --- a/data/models/anthropic_claude-opus-4-5.json +++ b/data/models/anthropic_claude-opus-4-5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4-5", "developer": "Anthropic", "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "5.59", - "total_run_cost": "558.51", - "average_steps": "41.07", - "percent_finished": "0.82" + "average_agent_cost": "22.76", + "total_run_cost": "2276.48", + "average_steps": "47.65", + "percent_finished": "0.77" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.7, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "22.76", - "total_run_cost": "2276.48", - "average_steps": "47.65", - "percent_finished": "0.77" + "average_agent_cost": "5.59", + "total_run_cost": "558.51", + "average_steps": "41.07", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.66, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "11.32", - "total_run_cost": "1132.47", - "average_steps": "21.99", - "percent_finished": "0.83" + "average_agent_cost": "13.08", + "total_run_cost": "1308.38", + "average_steps": "49.69", + "percent_finished": "0.74" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -227,34 +227,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.64, + "score": 0.49, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.43", - "total_run_cost": "343.32", - "average_steps": "20.06", - "percent_finished": "0.82" + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" } }, "generation_config": { @@ -282,7 +282,7 @@ } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.64, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "13.08", - "total_run_cost": "1308.38", - "average_steps": "49.69", - "percent_finished": "0.74" + "average_agent_cost": "3.43", + "total_run_cost": "343.32", + "average_steps": "20.06", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -387,9 +387,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "7.59", - "total_run_cost": "759.44", - "average_steps": "27.18", + "average_agent_cost": "6.3", + "total_run_cost": "630.56", + "average_steps": "24.16", "percent_finished": "1.0" } }, @@ -397,8 +397,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.5294, "uncertainty": { - "num_samples": 100 + "num_samples": 51 }, "details": { - "average_agent_cost": "6.3", - "total_run_cost": "630.56", - "average_steps": "24.16", - "percent_finished": "1.0" + "average_agent_cost": "11.66", + "total_run_cost": "594.68", + "average_steps": "31.04", + "percent_finished": "0.8431" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5294, + "score": 0.49, "uncertainty": { - "num_samples": 51 + "num_samples": 100 }, "details": { - "average_agent_cost": "11.66", - "total_run_cost": "594.68", - "average_steps": "31.04", - "percent_finished": "0.8431" + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.61, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "7.59", + "total_run_cost": "759.44", + "average_steps": "27.18", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -635,42 +635,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.61, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "11.32", + "total_run_cost": "1132.47", + "average_steps": "21.99", + "percent_finished": "0.83" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -682,15 +682,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -722,14 +722,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6061, + "score": 0.7423, "uncertainty": { - "num_samples": 99 + "num_samples": 97 }, "details": { - "average_agent_cost": "3.97", - "total_run_cost": "393.16", - "average_steps": "43.44", + "average_agent_cost": "5.6", + "total_run_cost": "543.62", + "average_steps": "31.76", "percent_finished": "1.0" } }, @@ -737,8 +737,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -750,8 +750,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -826,7 +826,7 @@ } }, { - "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -858,14 +858,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7423, + "score": 0.8072, "uncertainty": { - "num_samples": 97 + "num_samples": 83 }, "details": { - "average_agent_cost": "5.6", - "total_run_cost": "543.62", - "average_steps": "31.76", + "average_agent_cost": "2.96", + "total_run_cost": "245.78", + "average_steps": "34.1", "percent_finished": "1.0" } }, @@ -873,8 +873,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -886,15 +886,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -926,14 +926,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8072, + "score": 0.65, "uncertainty": { - "num_samples": 83 + "num_samples": 100 }, "details": { - "average_agent_cost": "2.96", - "total_run_cost": "245.78", - "average_steps": "34.1", + "average_agent_cost": "4.85", + "total_run_cost": "485.22", + "average_steps": "39.13", "percent_finished": "1.0" } }, @@ -941,8 +941,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -954,15 +954,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -994,14 +994,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.65, + "score": 0.6061, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "4.85", - "total_run_cost": "485.22", - "average_steps": "39.13", + "average_agent_cost": "3.97", + "total_run_cost": "393.16", + "average_steps": "43.44", "percent_finished": "1.0" } }, @@ -1009,8 +1009,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1022,8 +1022,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1098,7 +1098,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1130,14 +1130,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.74, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.72", - "total_run_cost": "36.55", - "average_steps": "12.22", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1145,8 +1145,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1158,15 +1158,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1198,14 +1198,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.72, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.78", - "total_run_cost": "39.67", - "average_steps": "11.88", + "average_agent_cost": "1.3", + "total_run_cost": "65.66", + "average_steps": "11.5", "percent_finished": "1.0" } }, @@ -1213,8 +1213,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1226,15 +1226,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1266,14 +1266,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.72, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "0.78", + "total_run_cost": "39.67", + "average_steps": "11.88", "percent_finished": "1.0" } }, @@ -1281,8 +1281,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1294,8 +1294,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1370,7 +1370,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1383,33 +1383,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.74, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "0.72", + "total_run_cost": "36.55", + "average_steps": "12.22", "percent_finished": "1.0" } }, @@ -1417,8 +1417,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1430,15 +1430,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1475,9 +1475,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "0.67", + "total_run_cost": "68.24", + "average_steps": "11.71", "percent_finished": "1.0" } }, @@ -1485,8 +1485,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1498,15 +1498,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1538,14 +1538,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.85, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.67", - "total_run_cost": "68.24", - "average_steps": "11.71", + "average_agent_cost": "0.55", + "total_run_cost": "56.18", + "average_steps": "12.54", "percent_finished": "1.0" } }, @@ -1553,8 +1553,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1566,15 +1566,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1606,14 +1606,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.85, + "score": 0.78, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "56.18", - "average_steps": "12.54", + "average_agent_cost": "0.47", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1621,8 +1621,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1634,15 +1634,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1655,33 +1655,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.78, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "65.66", - "average_steps": "11.5", + "average_agent_cost": "0.47", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1689,8 +1689,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1702,15 +1702,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1747,9 +1747,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "2.45", - "total_run_cost": "255.97", - "average_steps": "18.71", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1757,8 +1757,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1770,15 +1770,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1825,8 +1825,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1838,15 +1838,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1878,14 +1878,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.84, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "1.25", + "total_run_cost": "136.84", + "average_steps": "17.15", "percent_finished": "1.0" } }, @@ -1893,8 +1893,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1906,15 +1906,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1946,14 +1946,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.84, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.25", - "total_run_cost": "136.84", - "average_steps": "17.15", + "average_agent_cost": "1.06", + "total_run_cost": "114.62", + "average_steps": "13.77", "percent_finished": "1.0" } }, @@ -1961,8 +1961,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1974,15 +1974,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2014,14 +2014,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.06", - "total_run_cost": "114.62", - "average_steps": "13.77", + "average_agent_cost": "2.45", + "total_run_cost": "255.97", + "average_steps": "18.71", "percent_finished": "1.0" } }, @@ -2029,8 +2029,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2042,8 +2042,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } diff --git a/data/models/anthropic_claude-opus-4.1.json b/data/models/anthropic_claude-opus-4.1.json index dffd22cc4ffeffdda816bc746307e1e62ff77222..7cb0d9643500ef92a22ffb3a164b116b3c30b193 100644 --- a/data/models/anthropic_claude-opus-4.1.json +++ b/data/models/anthropic_claude-opus-4.1.json @@ -4,8 +4,8 @@ "id": "anthropic/claude-opus-4.1", "developer": "Anthropic", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 38.0, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 38.0, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.5.json b/data/models/anthropic_claude-opus-4.5.json index db911fda50bfaa4ee79d41805d7cf44e7a3058cc..8cd438cd6c457a625db5e35adee0adb0e1987648 100644 --- a/data/models/anthropic_claude-opus-4.5.json +++ b/data/models/anthropic_claude-opus-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "Goose", - "agent_organization": "Block" + "agent_name": "Claude Code", + "agent_organization": "Anthropic" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2026-01-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,11 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, - "uncertainty": { - "standard_error": { - "value": 2.4 - }, - "num_samples": 435 - } + "score": 51.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +64,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +78,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +102,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-12", + "evaluation_timestamp": "2025-11-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,11 +111,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7 + "score": 57.8, + "uncertainty": { + "standard_error": { + "value": 2.5 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -138,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -152,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -176,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-22", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -185,17 +185,11 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 58.4 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +206,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +220,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +244,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,11 +253,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4 + "score": 59.1, + "uncertainty": { + "standard_error": { + "value": 2.4 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -280,7 +280,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -368,7 +368,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -401,17 +401,17 @@ "max_score": 100.0 }, "score_details": { - "score": 63.1, + "score": 54.3, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -428,7 +428,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -442,7 +442,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -466,7 +466,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -475,17 +475,17 @@ "max_score": 100.0 }, "score_details": { - "score": 52.1, + "score": 63.1, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -502,7 +502,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -516,7 +516,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -540,7 +540,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -549,17 +549,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.3, + "score": 52.1, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -576,7 +576,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.6.json b/data/models/anthropic_claude-opus-4.6.json index 7886ec8ae8cad49aaa26033d604cb1ebebddeda0..39319e1a07ea2e5424bfa9b3da05363574f14160 100644 --- a/data/models/anthropic_claude-opus-4.6.json +++ b/data/models/anthropic_claude-opus-4.6.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.6", "developer": "Anthropic", "additional_details": { - "agent_name": "Droid", - "agent_organization": "Factory" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-07", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.0, + "score": 69.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,11 @@ "max_score": 100.0 }, "score_details": { - "score": 71.9, - "uncertainty": { - "standard_error": { - "value": 2.7 - }, - "num_samples": 435 - } + "score": 66.9 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-13", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +185,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, + "score": 66.5, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +212,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +226,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +250,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +259,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.7, + "score": 58.0, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +300,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +324,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-13", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +333,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.5, + "score": 74.7, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +398,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,11 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.9 + "score": 71.9, + "uncertainty": { + "standard_error": { + "value": 2.7 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 69.9, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-sonnet-4-20250514.json b/data/models/anthropic_claude-sonnet-4-20250514.json index 4792ff1b3b42b04e906eb39316e2865188e5cf63..b9ad033a7ea75f99c7213a0abf9a6fcf771340ca 100644 --- a/data/models/anthropic_claude-sonnet-4-20250514.json +++ b/data/models/anthropic_claude-sonnet-4-20250514.json @@ -9,8 +9,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -524,8 +524,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-sonnet-4-20250514/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-sonnet-4.5.json b/data/models/anthropic_claude-sonnet-4.5.json index 606e72f15a5601ecd2d0774a197b069b075f2ca1..84725fd141668d26191450ea056d65258afa8a61 100644 --- a/data/models/anthropic_claude-sonnet-4.5.json +++ b/data/models/anthropic_claude-sonnet-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-sonnet-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "CAMEL-AI", + "agent_organization": "CAMEL-AI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 46.5, + "score": 42.8, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.8, + "score": 40.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,11 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.7 + "score": 43.1, + "uncertainty": { + "standard_error": { + "value": 2.6 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.1, + "score": 42.5, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 40.1, + "score": 42.6, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -398,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -407,17 +413,11 @@ "max_score": 100.0 }, "score_details": { - "score": 42.5, - "uncertainty": { - "standard_error": { - "value": 2.8 - }, - "num_samples": 435 - } + "score": 42.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/camel-ai__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.6, + "score": 46.5, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CAMEL-AI\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/bunnycore_Llama-3.2-3B-Deep-Test.json b/data/models/bunnycore_Llama-3.2-3B-Deep-Test.json index 9eb6bc6b1e8f59d62f03d9d7ea00ea05b8b0e8c9..5ce0572be314ec7b607073f00ba87b7760238958 100644 --- a/data/models/bunnycore_Llama-3.2-3B-Deep-Test.json +++ b/data/models/bunnycore_Llama-3.2-3B-Deep-Test.json @@ -5,9 +5,9 @@ "developer": "bunnycore", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "3.607" + "params_billions": "1.803" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1775 + "score": 0.4652 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.295 + "score": 0.4531 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1284 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2643 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3647 + "score": 0.3394 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1049 + "score": 0.3152 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4652 + "score": 0.1775 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.295 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1284 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2643 + "score": 0.2517 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3394 + "score": 0.3647 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3152 + "score": 0.1049 } } ], diff --git a/data/models/cognitivecomputations_dolphin-2.9.2-Phi-3-Medium-abliterated.json b/data/models/cognitivecomputations_dolphin-2.9.2-Phi-3-Medium-abliterated.json index 4942452bc652e1b0662275861da63864a9a53556..5a05bbf781307c09b622fb50fc09f0f1a3fd17e9 100644 --- a/data/models/cognitivecomputations_dolphin-2.9.2-Phi-3-Medium-abliterated.json +++ b/data/models/cognitivecomputations_dolphin-2.9.2-Phi-3-Medium-abliterated.json @@ -5,7 +5,7 @@ "developer": "cognitivecomputations", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "13.96" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3613 + "score": 0.4124 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6123 + "score": 0.6383 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1239 + "score": 0.182 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.328 + "score": 0.3289 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4112 + "score": 0.4349 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4494 + "score": 0.4525 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4124 + "score": 0.3613 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6383 + "score": 0.6123 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.182 + "score": 0.1239 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.328 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4349 + "score": 0.4112 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4525 + "score": 0.4494 } } ], diff --git a/data/models/cpayne1303_llama-43m-beta.json b/data/models/cpayne1303_llama-43m-beta.json index 5243c9469aaa343a5b45890ec1189b4428db5a82..f4a13ffe702cda46b4208d4585e754b84fa898a6 100644 --- a/data/models/cpayne1303_llama-43m-beta.json +++ b/data/models/cpayne1303_llama-43m-beta.json @@ -5,7 +5,7 @@ "developer": "cpayne1303", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "0.043" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1949 + "score": 0.1916 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2965 + "score": 0.2977 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0045 + "score": 0.0 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3885 + "score": 0.3872 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1111 + "score": 0.1132 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1916 + "score": 0.1949 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2965 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0045 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3872 + "score": 0.3885 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1132 + "score": 0.1111 } } ], diff --git a/data/models/deepseek_deepseek-v3.1.json b/data/models/deepseek_deepseek-v3.1.json index 28271d97eb42a16e44db35269bdfee0d33d2475a..52e144fa11c26943b339cba2c9fa28b1d9098c61 100644 --- a/data/models/deepseek_deepseek-v3.1.json +++ b/data/models/deepseek_deepseek-v3.1.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/fblgit_TheBeagle-v2beta-32B-MGS.json b/data/models/fblgit_TheBeagle-v2beta-32B-MGS.json index be17afdcfffac7cd8267c81de014533270b46d3f..57e8a0c9c4547329f51827d27aa7733799780c1a 100644 --- a/data/models/fblgit_TheBeagle-v2beta-32B-MGS.json +++ b/data/models/fblgit_TheBeagle-v2beta-32B-MGS.json @@ -5,7 +5,7 @@ "developer": "fblgit", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5181 + "score": 0.4503 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7033 + "score": 0.7035 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4947 + "score": 0.3943 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3826 + "score": 0.401 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5008 + "score": 0.5021 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5915 + "score": 0.5911 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4503 + "score": 0.5181 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7035 + "score": 0.7033 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.4947 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.401 + "score": 0.3826 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5021 + "score": 0.5008 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5911 + "score": 0.5915 } } ], diff --git a/data/models/google_Gemini_3_Flash.json b/data/models/google_Gemini_3_Flash.json index a61c04322fdc41b3a4e6f1f4da89287f5d95639a..f3dcb98222f3a45bbec3325acba502991a42fcb5 100644 --- a/data/models/google_Gemini_3_Flash.json +++ b/data/models/google_Gemini_3_Flash.json @@ -6,6 +6,53 @@ "inference_platform": "unknown" }, "evaluations": [ + { + "evaluation_id": "ace/google_gemini-3-flash/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.415 + }, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + }, { "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", @@ -205,53 +252,6 @@ } } }, - { - "evaluation_id": "ace/google_gemini-3-flash/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.415 - }, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - }, { "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/google_flan-t5-xl.json b/data/models/google_flan-t5-xl.json index 7dd94c33f5fee062e6373821c397609d90d6c2ae..cdfcb6cc4c8ade24fd752915b01166ac7893b099 100644 --- a/data/models/google_flan-t5-xl.json +++ b/data/models/google_flan-t5-xl.json @@ -5,7 +5,7 @@ "developer": "google", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "T5ForConditionalGeneration", "params_billions": "2.85" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2237 + "score": 0.2207 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.4537 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0008 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2525 + "score": 0.2458 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4181 + "score": 0.422 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2147 + "score": 0.2142 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2207 + "score": 0.2237 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4537 + "score": 0.4531 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0008 + "score": 0.0076 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2458 + "score": 0.2525 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.422 + "score": 0.4181 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2142 + "score": 0.2147 } } ], diff --git a/data/models/google_gemini-2.5-flash.json b/data/models/google_gemini-2.5-flash.json index 100ef3ccb6413e0e23161c5f90be16667820a8bb..a993afbf9c67f237669586a95c4346d7b5bd221c 100644 --- a/data/models/google_gemini-2.5-flash.json +++ b/data/models/google_gemini-2.5-flash.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-flash/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,7 +1376,7 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 16.9, "uncertainty": { "standard_error": { "value": 2.4 @@ -1386,7 +1386,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,7 +1450,7 @@ "max_score": 100.0 }, "score_details": { - "score": 16.9, + "score": 16.4, "uncertainty": { "standard_error": { "value": 2.4 @@ -1460,7 +1460,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-2.5-pro.json b/data/models/google_gemini-2.5-pro.json index c78a20c000adf11c0318e63011a9dba7982505e7..27a9d48ea316849f1130e704e84871fca27b4413 100644 --- a/data/models/google_gemini-2.5-pro.json +++ b/data/models/google_gemini-2.5-pro.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-2.5-pro/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.1, + "score": 19.6, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 19.6, + "score": 26.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-flash.json b/data/models/google_gemini-3-flash.json index a020edf7d2de6c31874320ca6b56508a5cced539..6f45404e1daa03a1dd984d9d83a1483a4ef2b12d 100644 --- a/data/models/google_gemini-3-flash.json +++ b/data/models/google_gemini-3-flash.json @@ -10,7 +10,7 @@ }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.3, + "score": 51.7, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-07", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7, + "score": 64.3, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-pro-preview.json b/data/models/google_gemini-3-pro-preview.json index 72d051735ad53e09d24aa98fe02cfc581a1e9522..d5eb006f74d93ee8fec5c4b9e19e6acaccb108a0 100644 --- a/data/models/google_gemini-3-pro-preview.json +++ b/data/models/google_gemini-3-pro-preview.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro-preview", "developer": "Google", "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.36, + "score": 0.505, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.11", - "total_run_cost": "310.55", - "average_steps": "38.01", - "percent_finished": "0.86" + "average_agent_cost": "1.88", + "total_run_cost": "188.19", + "average_steps": "21.76", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.13, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.54", - "total_run_cost": "254.25", - "average_steps": "49.13", - "percent_finished": "0.71" + "average_agent_cost": "1.3", + "total_run_cost": "130.49", + "average_steps": "22.59", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.13, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "130.49", - "average_steps": "22.59", - "percent_finished": "1.0" + "average_agent_cost": "2.54", + "total_run_cost": "254.25", + "average_steps": "49.13", + "percent_finished": "0.71" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -206,8 +206,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -282,7 +282,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -295,42 +295,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.505, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.88", - "total_run_cost": "188.19", - "average_steps": "21.76", - "percent_finished": "0.99" + "average_agent_cost": "2.85", + "total_run_cost": "284.68", + "average_steps": "22.88", + "percent_finished": "0.7" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -363,42 +363,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.36, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.39", - "total_run_cost": "239.0", - "average_steps": "29.63", - "percent_finished": "0.69" + "average_agent_cost": "3.11", + "total_run_cost": "310.55", + "average_steps": "38.01", + "percent_finished": "0.86" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.57, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.85", - "total_run_cost": "284.68", - "average_steps": "22.88", - "percent_finished": "0.7" + "average_agent_cost": "2.39", + "total_run_cost": "239.0", + "average_steps": "29.63", + "percent_finished": "0.69" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3333, + "score": 0.48, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.64", - "total_run_cost": "63.79", - "average_steps": "8.45", - "percent_finished": "0.6061" + "average_agent_cost": "0.44", + "total_run_cost": "44.18", + "average_steps": "7.85", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.3333, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.44", - "total_run_cost": "44.18", - "average_steps": "7.85", - "percent_finished": "0.99" + "average_agent_cost": "0.64", + "total_run_cost": "63.79", + "average_steps": "8.45", + "percent_finished": "0.6061" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -682,16 +682,16 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "global-mmlu-lite/google_gemini-3-pro-preview/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemini-3-pro-preview/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1205,8 +1205,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemini-3-pro-preview/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemini-3-pro-preview/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -1924,7 +1924,7 @@ } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1956,14 +1956,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.7234, "uncertainty": { - "num_samples": 100 + "num_samples": 94 }, "details": { - "average_agent_cost": "0.7", - "total_run_cost": "69.56", - "average_steps": "32.55", + "average_agent_cost": "1.58", + "total_run_cost": "148.44", + "average_steps": "32.36", "percent_finished": "1.0" } }, @@ -1971,8 +1971,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1984,15 +1984,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2024,14 +2024,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7234, + "score": 0.71, "uncertainty": { - "num_samples": 94 + "num_samples": 100 }, "details": { - "average_agent_cost": "1.58", - "total_run_cost": "148.44", - "average_steps": "32.36", + "average_agent_cost": "0.7", + "total_run_cost": "69.56", + "average_steps": "32.55", "percent_finished": "1.0" } }, @@ -2039,8 +2039,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2052,15 +2052,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2097,9 +2097,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "0.34", - "total_run_cost": "17.45", - "average_steps": "12.62", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2107,8 +2107,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2120,15 +2120,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2160,14 +2160,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.62, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.18", - "average_steps": "10.9", + "average_agent_cost": "0.34", + "total_run_cost": "17.45", + "average_steps": "12.62", "percent_finished": "1.0" } }, @@ -2175,8 +2175,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2188,15 +2188,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2228,14 +2228,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.68, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.2", + "total_run_cost": "10.29", + "average_steps": "12.28", "percent_finished": "1.0" } }, @@ -2243,8 +2243,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2256,15 +2256,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2296,14 +2296,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.62, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.21", + "total_run_cost": "11.18", + "average_steps": "10.9", "percent_finished": "1.0" } }, @@ -2311,8 +2311,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2324,15 +2324,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2345,33 +2345,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.7, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "21.43", - "average_steps": "11.3", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2379,8 +2379,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2392,15 +2392,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2413,33 +2413,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.7576, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.2", - "total_run_cost": "10.29", - "average_steps": "12.28", + "average_agent_cost": "0.21", + "total_run_cost": "21.43", + "average_steps": "11.3", "percent_finished": "1.0" } }, @@ -2536,7 +2536,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2568,14 +2568,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.82, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "16.64", - "average_steps": "11.25", + "average_agent_cost": "0.27", + "total_run_cost": "27.48", + "average_steps": "10.62", "percent_finished": "1.0" } }, @@ -2583,8 +2583,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2596,15 +2596,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2636,14 +2636,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.82, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.27", - "total_run_cost": "27.48", - "average_steps": "10.62", + "average_agent_cost": "0.16", + "total_run_cost": "16.64", + "average_steps": "11.25", "percent_finished": "1.0" } }, @@ -2651,8 +2651,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2664,8 +2664,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2740,7 +2740,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2772,14 +2772,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.88, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", + "average_agent_cost": "0.35", + "total_run_cost": "40.25", + "average_steps": "12.71", "percent_finished": "1.0" } }, @@ -2787,8 +2787,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2800,15 +2800,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2840,23 +2840,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8876, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.54", - "total_run_cost": "58.29", - "average_steps": "10.82", - "percent_finished": "0.89" + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2868,15 +2868,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2908,23 +2908,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.88, + "score": 0.8876, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.35", - "total_run_cost": "40.25", - "average_steps": "12.71", - "percent_finished": "1.0" + "average_agent_cost": "0.54", + "total_run_cost": "58.29", + "average_steps": "10.82", + "percent_finished": "0.89" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2936,8 +2936,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -3012,7 +3012,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -3059,8 +3059,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -3072,8 +3072,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } diff --git a/data/models/google_gemini-3-pro.json b/data/models/google_gemini-3-pro.json index eae93e48214f42eb08f89532b3b13d405fb3768b..e909b260cb8d04bf882146b1c3851eca1bd7d4f7 100644 --- a/data/models/google_gemini-3-pro.json +++ b/data/models/google_gemini-3-pro.json @@ -4,8 +4,8 @@ "id": "google/gemini-3-pro", "developer": "Google", "additional_details": { - "agent_name": "CodeBrain-1", - "agent_organization": "Feeling AI" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.1, + "score": 56.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 61.8, + "score": 61.1, "uncertainty": { "standard_error": { "value": 2.8 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.0, + "score": 62.2, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-21", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.9, + "score": 61.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -454,7 +454,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -478,7 +478,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2025-11-21", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -487,17 +487,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.2, + "score": 56.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -514,7 +514,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3.1-pro.json b/data/models/google_gemini-3.1-pro.json index 0e5f8fe75691718cffdd5684a2f8910d846de44c..945834533f47e11456a671dc36fdab03175c7bd2 100644 --- a/data/models/google_gemini-3.1-pro.json +++ b/data/models/google_gemini-3.1-pro.json @@ -4,13 +4,13 @@ "id": "google/gemini-3.1-pro", "developer": "Google", "additional_details": { - "agent_name": "Forge Code", - "agent_organization": "Forge Code" + "agent_name": "Terminus-KIRA", + "agent_organization": "KRAFTON AI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-03-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.8, + "score": 78.4, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 1.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/forge-code__gemini-3.1-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__gemini-3.1-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-02", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 78.4, + "score": 74.8, "uncertainty": { "standard_error": { - "value": 1.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Forge Code\" -m \"Gemini 3.1 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Gemini 3.1 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemma-2-9b-it.json b/data/models/google_gemma-2-9b-it.json index 4a4b375fb8b794d17a16dc51702642587f998849..98170ef9eb836f3f1ddfe0b38123efa76c67c5cc 100644 --- a/data/models/google_gemma-2-9b-it.json +++ b/data/models/google_gemma-2-9b-it.json @@ -1,14 +1,8 @@ { "model_info": { - "name": "gemma-2-9b-it", + "name": "Gemma 2 9B Instruct", "id": "google/gemma-2-9b-it", - "developer": "google", - "inference_platform": "unknown", - "additional_details": { - "precision": "bfloat16", - "architecture": "Gemma2ForCausalLM", - "params_billions": "9.0" - } + "developer": "unknown" }, "evaluations": [ { @@ -512,6 +506,46 @@ ], "detailed_evaluation_results": null, "generation_config": null + }, + { + "evaluation_id": "la_leaderboard/google/gemma-2-9b-it/1774451270", + "retrieved_timestamp": "2024-10-27T00:00:00Z", + "source_metadata": { + "source_name": "La Leaderboard", + "source_type": "evaluation_run", + "source_url": "https://huggingface.co/spaces/la-leaderboard/la-leaderboard", + "source_organization_name": "La Leaderboard", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "custom", + "version": "1.0" + }, + "benchmark": "la_leaderboard", + "evaluation_results": [ + { + "evaluation_name": "la_leaderboard", + "metric_config": { + "evaluation_description": "La Leaderboard: LLM evaluation for Spanish varieties and languages of Spain and Latin America", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 100 + }, + "score_details": { + "score": 33.62 + }, + "source_data": { + "source_type": "url", + "dataset_name": "La Leaderboard composite dataset", + "url": [ + "https://huggingface.co/spaces/la-leaderboard/la-leaderboard" + ] + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null } ] } \ No newline at end of file diff --git a/data/models/hendrydong_Mistral-RM-for-RAFT-GSHF-v0.json b/data/models/hendrydong_Mistral-RM-for-RAFT-GSHF-v0.json index 357f438b07831c53712dd63f870b6b2401c4d681..69bdc21b5b14ca68b6163472e8eef6fd89e268f5 100644 --- a/data/models/hendrydong_Mistral-RM-for-RAFT-GSHF-v0.json +++ b/data/models/hendrydong_Mistral-RM-for-RAFT-GSHF-v0.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5851 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5779 + "score": 0.7847 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.9832 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6011 + "score": 0.5789 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6956 + "score": 0.85 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6747 + "score": 0.7434 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5988 + "score": 0.7508 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", + "evaluation_id": "reward-bench-2/hendrydong_Mistral-RM-for-RAFT-GSHF-v0/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7847 + "score": 0.5851 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9832 + "score": 0.5779 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5789 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6011 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.85 + "score": 0.6956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7434 + "score": 0.6747 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7508 + "score": 0.5988 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/infly_INF-ORM-Llama3.1-70B.json b/data/models/infly_INF-ORM-Llama3.1-70B.json index e7947ee940015eb0652da9a52891a9ab47739595..82e76ad6cd43b3a105b801967d5f616a3924844a 100644 --- a/data/models/infly_INF-ORM-Llama3.1-70B.json +++ b/data/models/infly_INF-ORM-Llama3.1-70B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7648 + "score": 0.9511 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7411 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9101 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.9365 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9644 + "score": 0.9912 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.903 + "score": 0.7648 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8622 + "score": 0.7411 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9511 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9101 + "score": 0.9644 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9365 + "score": 0.903 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9912 + "score": 0.8622 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/internlm_internlm2-1_8b-reward.json b/data/models/internlm_internlm2-1_8b-reward.json index db02a96edd7d4deec66c9db68ed23c5ecfdc96f9..fdd95af043dae0ccf6943006b6d82bc9a69847ba 100644 --- a/data/models/internlm_internlm2-1_8b-reward.json +++ b/data/models/internlm_internlm2-1_8b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8217 + "score": 0.3902 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9358 + "score": 0.2758 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6623 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.4426 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8724 + "score": 0.4711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3902 + "score": 0.596 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2758 + "score": 0.1934 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.8217 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4426 + "score": 0.9358 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4711 + "score": 0.6623 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.596 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1934 + "score": 0.8724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/internlm_internlm2-20b-reward.json b/data/models/internlm_internlm2-20b-reward.json index 4de166855af75438b275f87369424e25bd91920e..db57bc6ddd293d585b6bca7ac06b0b270dabb864 100644 --- a/data/models/internlm_internlm2-20b-reward.json +++ b/data/models/internlm_internlm2-20b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9016 + "score": 0.5628 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9888 + "score": 0.5558 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7654 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8946 + "score": 0.5738 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9576 + "score": 0.6111 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-20b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5628 + "score": 0.7253 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5558 + "score": 0.5483 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-20b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.9016 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5738 + "score": 0.9888 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6111 + "score": 0.7654 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7253 + "score": 0.8946 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5483 + "score": 0.9576 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/jaspionjader_Kosmos-EVAA-Fusion-8B.json b/data/models/jaspionjader_Kosmos-EVAA-Fusion-8B.json index 912256e1b8a2fead4aedc10bc547740b2edcb38a..ba928fd33c286ceb6a74c1a7ec1b4a2774b1f177 100644 --- a/data/models/jaspionjader_Kosmos-EVAA-Fusion-8B.json +++ b/data/models/jaspionjader_Kosmos-EVAA-Fusion-8B.json @@ -5,7 +5,7 @@ "developer": "jaspionjader", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4418 + "score": 0.4345 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.5419 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1352 + "score": 0.1292 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3087 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3854 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4345 + "score": 0.4418 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5419 + "score": 0.5406 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1292 + "score": 0.1352 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3062 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3854 + "score": 0.386 } } ], diff --git a/data/models/meta-llama_Meta-Llama-3.1-8B-Instruct.json b/data/models/meta-llama_Meta-Llama-3.1-8B-Instruct.json new file mode 100644 index 0000000000000000000000000000000000000000..2e86cfb6a810e6b1d7164056a5ea21537f5bb762 --- /dev/null +++ b/data/models/meta-llama_Meta-Llama-3.1-8B-Instruct.json @@ -0,0 +1,49 @@ +{ + "model_info": { + "name": "Meta Llama 3.1 8B Instruct", + "id": "meta-llama/Meta-Llama-3.1-8B-Instruct", + "developer": "unknown" + }, + "evaluations": [ + { + "evaluation_id": "la_leaderboard/meta-llama/Meta-Llama-3.1-8B-Instruct/1774451270", + "retrieved_timestamp": "2024-10-27T00:00:00Z", + "source_metadata": { + "source_name": "La Leaderboard", + "source_type": "evaluation_run", + "source_url": "https://huggingface.co/spaces/la-leaderboard/la-leaderboard", + "source_organization_name": "La Leaderboard", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "custom", + "version": "1.0" + }, + "benchmark": "la_leaderboard", + "evaluation_results": [ + { + "evaluation_name": "la_leaderboard", + "metric_config": { + "evaluation_description": "La Leaderboard: LLM evaluation for Spanish varieties and languages of Spain and Latin America", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 100 + }, + "score_details": { + "score": 30.23 + }, + "source_data": { + "source_type": "url", + "dataset_name": "La Leaderboard composite dataset", + "url": [ + "https://huggingface.co/spaces/la-leaderboard/la-leaderboard" + ] + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + } + ] +} \ No newline at end of file diff --git a/data/models/meta-llama_Meta-Llama-3.1-8B.json b/data/models/meta-llama_Meta-Llama-3.1-8B.json new file mode 100644 index 0000000000000000000000000000000000000000..e0736feccce1fe2ca3631c5893d8bd446ee1f427 --- /dev/null +++ b/data/models/meta-llama_Meta-Llama-3.1-8B.json @@ -0,0 +1,49 @@ +{ + "model_info": { + "name": "Meta Llama 3.1 8B", + "id": "meta-llama/Meta-Llama-3.1-8B", + "developer": "unknown" + }, + "evaluations": [ + { + "evaluation_id": "la_leaderboard/meta-llama/Meta-Llama-3.1-8B/1774451270", + "retrieved_timestamp": "2024-10-27T00:00:00Z", + "source_metadata": { + "source_name": "La Leaderboard", + "source_type": "evaluation_run", + "source_url": "https://huggingface.co/spaces/la-leaderboard/la-leaderboard", + "source_organization_name": "La Leaderboard", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "custom", + "version": "1.0" + }, + "benchmark": "la_leaderboard", + "evaluation_results": [ + { + "evaluation_name": "la_leaderboard", + "metric_config": { + "evaluation_description": "La Leaderboard: LLM evaluation for Spanish varieties and languages of Spain and Latin America", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 100 + }, + "score_details": { + "score": 27.04 + }, + "source_data": { + "source_type": "url", + "dataset_name": "La Leaderboard composite dataset", + "url": [ + "https://huggingface.co/spaces/la-leaderboard/la-leaderboard" + ] + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + } + ] +} \ No newline at end of file diff --git a/data/models/microsoft_phi-4.json b/data/models/microsoft_phi-4.json index b963af9da3550d27dbd1c016591354645798b294..97f523bcb3123f3c15262b578f3a90bbfd2182ab 100644 --- a/data/models/microsoft_phi-4.json +++ b/data/models/microsoft_phi-4.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Phi3ForCausalLM", "params_billions": "14.66" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0488 + "score": 0.0585 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6703 + "score": 0.6691 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2787 + "score": 0.3165 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.401 + "score": 0.406 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5295 + "score": 0.5287 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0585 + "score": 0.0488 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6691 + "score": 0.6703 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3165 + "score": 0.2787 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.406 + "score": 0.401 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5287 + "score": 0.5295 } } ], diff --git a/data/models/migtissera_Trinity-2-Codestral-22B-v0.2.json b/data/models/migtissera_Trinity-2-Codestral-22B-v0.2.json index e73a7af4d30076cd08273bf7149a3e040e29d637..679817b0f46278305e51ca89bfbcb28c81768bf7 100644 --- a/data/models/migtissera_Trinity-2-Codestral-22B-v0.2.json +++ b/data/models/migtissera_Trinity-2-Codestral-22B-v0.2.json @@ -5,7 +5,7 @@ "developer": "migtissera", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "22.247" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4345 + "score": 0.443 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5686 + "score": 0.5706 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0838 + "score": 0.0869 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3003 + "score": 0.3079 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4045 + "score": 0.4031 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.334 + "score": 0.3354 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.443 + "score": 0.4345 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5706 + "score": 0.5686 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0869 + "score": 0.0838 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3079 + "score": 0.3003 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4031 + "score": 0.4045 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.334 } } ], diff --git a/data/models/minimax_minimax-m2.1.json b/data/models/minimax_minimax-m2.1.json index 5402858eec966e47dd955cabd469c55c7620df55..81e83c79cbaf0b76263df086ba1f1fad96eacfe5 100644 --- a/data/models/minimax_minimax-m2.1.json +++ b/data/models/minimax_minimax-m2.1.json @@ -4,13 +4,13 @@ "id": "minimax/minimax-m2.1", "developer": "MiniMax", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Crux", + "agent_organization": "Roam" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-22", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 36.6, + "score": 29.2, "uncertainty": { "standard_error": { "value": 2.9 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 36.6, "uncertainty": { "standard_error": { "value": 2.9 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/mistralai_Mixtral-8x7B-v0.1.json b/data/models/mistralai_Mixtral-8x7B-v0.1.json index 050085c95f369d687bc49b467919c5816819a254..8cc966516ff05e97a419392beb962c7e43051d10 100644 --- a/data/models/mistralai_Mixtral-8x7B-v0.1.json +++ b/data/models/mistralai_Mixtral-8x7B-v0.1.json @@ -5,7 +5,7 @@ "developer": "mistralai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MixtralForCausalLM", "params_billions": "46.703" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2415 + "score": 0.2326 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5087 + "score": 0.5098 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.102 + "score": 0.0937 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3205 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4321 + "score": 0.4413 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.385 + "score": 0.3871 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2326 + "score": 0.2415 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5087 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.102 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3205 + "score": 0.3138 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4413 + "score": 0.4321 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3871 + "score": 0.385 } } ], diff --git a/data/models/mistralai_mistral-medium-3.json b/data/models/mistralai_mistral-medium-3.json index 8c2cd1ffe5583f8e08ea6a503148c17defbde8f4..3d0755e0237b9760231fee380b173b29d57eab4a 100644 --- a/data/models/mistralai_mistral-medium-3.json +++ b/data/models/mistralai_mistral-medium-3.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/mlabonne_NeuralDaredevil-8B-abliterated.json b/data/models/mlabonne_NeuralDaredevil-8B-abliterated.json index 3aa1261bc3f82b0f0c39c10cc6fa188b2dbae41f..bf81ba9fbed344ef29adc82aada2da4630eab1e8 100644 --- a/data/models/mlabonne_NeuralDaredevil-8B-abliterated.json +++ b/data/models/mlabonne_NeuralDaredevil-8B-abliterated.json @@ -5,7 +5,7 @@ "developer": "mlabonne", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7561 + "score": 0.4162 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5111 + "score": 0.5124 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0906 + "score": 0.0853 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3029 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.415 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3841 + "score": 0.3802 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.7561 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5124 + "score": 0.5111 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0853 + "score": 0.0906 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3029 + "score": 0.3062 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.415 + "score": 0.4019 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3841 } } ], diff --git a/data/models/multiple_multiple.json b/data/models/multiple_multiple.json index 962582d1837ba83db0927dd438888b7d933863ce..2ac193c3f6251d7c94a588973f2a9ae646a569ca 100644 --- a/data/models/multiple_multiple.json +++ b/data/models/multiple_multiple.json @@ -4,13 +4,13 @@ "id": "multiple/multiple", "developer": "Multiple", "additional_details": { - "agent_name": "Warp", - "agent_organization": "Warp" + "agent_name": "OB-1", + "agent_organization": "OpenBlock Labs" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-20", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 58.4, "uncertainty": { "standard_error": { "value": 2.8 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4, + "score": 61.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-11-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,10 +265,10 @@ "max_score": 100.0 }, "score_details": { - "score": 61.2, + "score": 50.1, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.7 }, "num_samples": 435 } @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-05", + "evaluation_timestamp": "2025-11-20", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 72.4, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-11", + "evaluation_timestamp": "2026-03-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 50.1, + "score": 72.4, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/nicolinho_QRM-Gemma-2-27B.json b/data/models/nicolinho_QRM-Gemma-2-27B.json index 98185886d3c230ddcd90456c69a5aeed49795fc5..1dea90f885df9d34139a9ef21e55b3dcce1a25fd 100644 --- a/data/models/nicolinho_QRM-Gemma-2-27B.json +++ b/data/models/nicolinho_QRM-Gemma-2-27B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.9444 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7853 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3719 + "score": 0.9013 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.927 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9578 + "score": 0.9826 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9535 + "score": 0.7667 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8321 + "score": 0.7853 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9444 + "score": 0.3719 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9013 + "score": 0.9578 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.927 + "score": 0.9535 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9826 + "score": 0.8321 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/nicolinho_QRM-Llama3.1-8B-v2.json b/data/models/nicolinho_QRM-Llama3.1-8B-v2.json index 0df8878cad15f33ea391f78f6a5406e147177591..71e586c5d191f366e8b76150e58f0f9807a69f6b 100644 --- a/data/models/nicolinho_QRM-Llama3.1-8B-v2.json +++ b/data/models/nicolinho_QRM-Llama3.1-8B-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7074 + "score": 0.9314 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6653 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4062 + "score": 0.8684 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.612 + "score": 0.9257 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9467 + "score": 0.9677 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8909 + "score": 0.7074 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7234 + "score": 0.6653 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9314 + "score": 0.4062 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.612 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8684 + "score": 0.9467 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9257 + "score": 0.8909 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9677 + "score": 0.7234 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/nisten_franqwenstein-35b.json b/data/models/nisten_franqwenstein-35b.json index cafe04278fd17057e0ad645aed2cef746b5e0bdd..67b39cd32d9a18463f66f47958bb5c24597bcb43 100644 --- a/data/models/nisten_franqwenstein-35b.json +++ b/data/models/nisten_franqwenstein-35b.json @@ -5,7 +5,7 @@ "developer": "nisten", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "34.714" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3799 + "score": 0.3914 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6647 + "score": 0.6591 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3406 + "score": 0.3044 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4035 + "score": 0.3591 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.494 + "score": 0.4681 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5731 + "score": 0.5611 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3914 + "score": 0.3799 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6591 + "score": 0.6647 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3044 + "score": 0.3406 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3591 + "score": 0.4035 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4681 + "score": 0.494 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5611 + "score": 0.5731 } } ], diff --git a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json index ac66670d2b371f198d2e8326c041663a689afe21..d58f793ad4b7123313c6b2627331c19f8f6aa9ed 100644 --- a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json +++ b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json @@ -5,7 +5,7 @@ "developer": "ontocord", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.759" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1162 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3171 + "score": 0.3184 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0113 + "score": 0.0076 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2685 + "score": 0.2634 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.3447 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1124 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1162 + "score": 0.1128 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3184 + "score": 0.3171 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0113 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2634 + "score": 0.2685 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3447 + "score": 0.346 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1124 + "score": 0.1129 } } ], diff --git a/data/models/oopere_Llama-FinSent-S.json b/data/models/oopere_Llama-FinSent-S.json index 4304bcd53168ed4fd51adf4213bcd9660265d629..547a7ef41580d39906f8398271f37052ca4bd621 100644 --- a/data/models/oopere_Llama-FinSent-S.json +++ b/data/models/oopere_Llama-FinSent-S.json @@ -5,7 +5,7 @@ "developer": "oopere", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "0.914" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2119 + "score": 0.2164 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3156 + "score": 0.3169 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0181 + "score": 0.0128 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2567 + "score": 0.2584 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.113 + "score": 0.1134 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2164 + "score": 0.2119 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3169 + "score": 0.3156 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0128 + "score": 0.0181 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2584 + "score": 0.2567 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1134 + "score": 0.113 } } ], diff --git a/data/models/openai_gpt-4o-mini-2024-07-18.json b/data/models/openai_gpt-4o-mini-2024-07-18.json index 34e2570b442f377162a4ef5d228d2765d94a549e..0d0bfaf896189ad5d882f10966ca4ebf67ca8152 100644 --- a/data/models/openai_gpt-4o-mini-2024-07-18.json +++ b/data/models/openai_gpt-4o-mini-2024-07-18.json @@ -4,7 +4,7 @@ "id": "openai/gpt-4o-mini-2024-07-18", "developer": "openai", "additional_details": { - "model_type": "Generative" + "model_type": "Generative RM" } }, "evaluations": [ @@ -2126,10 +2126,10 @@ } }, { - "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2148,104 +2148,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5796 + "score": 0.8007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4105 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.6075 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.8081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8374 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7414 + "score": 0.5796 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2254,135 +2278,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6962 + "score": 0.4105 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8007 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6075 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8081 + "score": 0.7414 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8374 + "score": 0.6962 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openai_gpt-5-2025-08-07.json b/data/models/openai_gpt-5-2025-08-07.json index 73a59bb6a004f262dd12f38906a946bc9718abbf..724f752582fce96f498cf28c110d297d1f461c38 100644 --- a/data/models/openai_gpt-5-2025-08-07.json +++ b/data/models/openai_gpt-5-2025-08-07.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/openai_gpt-5-2025-08-07/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/openai_gpt-5-mini.json b/data/models/openai_gpt-5-mini.json index b1410fe13d2501fc3f89237d9e985bed25d655cf..aeda61e4e83dad7b8cfd4764765996bc5fcb266f 100644 --- a/data/models/openai_gpt-5-mini.json +++ b/data/models/openai_gpt-5-mini.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-mini", "developer": "OpenAI", "additional_details": { - "agent_name": "Codex CLI", - "agent_organization": "OpenAI" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 24.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 22.2, + "score": 31.9, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 24.0, + "score": 22.2, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 31.9, + "score": 29.2, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-nano.json b/data/models/openai_gpt-5-nano.json index e4e8ae16cf51db0ba1405657d05afdf54cd350ed..3d33e8721b90a5dfdec90a430cbe292d4f08b47d 100644 --- a/data/models/openai_gpt-5-nano.json +++ b/data/models/openai_gpt-5-nano.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-nano", "developer": "OpenAI", "additional_details": { - "agent_name": "Mini-SWE-Agent", - "agent_organization": "Princeton" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 9.9, + "score": 11.5, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 11.5, + "score": 9.9, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.9, + "score": 7.0, "uncertainty": { "standard_error": { "value": 1.9 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.0, + "score": 7.9, "uncertainty": { "standard_error": { "value": 1.9 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.1-codex.json b/data/models/openai_gpt-5.1-codex.json index 79527746be168971f52c351c05aaeb3b5b9ecf66..97273068694d9637b3c058ef180958a58e1898d8 100644 --- a/data/models/openai_gpt-5.1-codex.json +++ b/data/models/openai_gpt-5.1-codex.json @@ -4,8 +4,8 @@ "id": "openai/gpt-5.1-codex", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Letta Code", + "agent_organization": "Letta" } }, "evaluations": [ @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 53.5, + "score": 36.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-17", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 36.9, + "score": 53.5, "uncertainty": { "standard_error": { - "value": 3.2 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.2-2025-12-11.json b/data/models/openai_gpt-5.2-2025-12-11.json index 0e67f7333253f410afc4237865ce72da2dc4610f..8a3e2f6d5d079f66e64fa66d652280a63cd7c3ac 100644 --- a/data/models/openai_gpt-5.2-2025-12-11.json +++ b/data/models/openai_gpt-5.2-2025-12-11.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2-2025-12-11", "developer": "OpenAI", "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.071, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.55", + "total_run_cost": "55.03", + "average_steps": "51.59", + "percent_finished": "0.61" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -125,8 +125,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.22, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.36", - "total_run_cost": "36.37", - "average_steps": "10.05", - "percent_finished": "1.0" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -261,8 +261,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.071, + "score": 0.22, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "55.03", - "average_steps": "51.59", - "percent_finished": "0.61" + "average_agent_cost": "0.36", + "total_run_cost": "36.37", + "average_steps": "10.05", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -342,15 +342,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,23 +382,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.46, + "score": 0.43, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "29.78", - "average_steps": "8.14", - "percent_finished": "0.99" + "average_agent_cost": "0.43", + "total_run_cost": "43.11", + "average_steps": "8.97", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.46, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.38", - "total_run_cost": "38.21", - "average_steps": "14.27", - "percent_finished": "1.0" + "average_agent_cost": "0.3", + "total_run_cost": "29.78", + "average_steps": "8.14", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.43, + "score": 0.26, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.43", - "total_run_cost": "43.11", - "average_steps": "8.97", - "percent_finished": "1.0" + "average_agent_cost": "0.17", + "total_run_cost": "17.31", + "average_steps": "6.57", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -546,15 +546,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,23 +586,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.26, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.17", - "total_run_cost": "17.31", - "average_steps": "6.57", - "percent_finished": "0.99" + "average_agent_cost": "0.38", + "total_run_cost": "38.21", + "average_steps": "14.27", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -614,8 +614,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -905,7 +905,7 @@ } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -937,14 +937,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5455, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.26", + "total_run_cost": "25.64", + "average_steps": "20.44", "percent_finished": "1.0" } }, @@ -952,8 +952,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -965,15 +965,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1005,14 +1005,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5455, + "score": 0.57, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.26", - "total_run_cost": "25.64", - "average_steps": "20.44", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -1020,8 +1020,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1033,15 +1033,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1054,33 +1054,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.5253, "uncertainty": { - "num_samples": 50 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.45", + "total_run_cost": "44.58", + "average_steps": "19.98", "percent_finished": "1.0" } }, @@ -1088,8 +1088,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1101,15 +1101,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1141,14 +1141,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "5.77", - "average_steps": "11.4", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1156,8 +1156,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1169,8 +1169,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1245,7 +1245,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1277,14 +1277,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.29", - "total_run_cost": "15.28", - "average_steps": "10.68", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1292,8 +1292,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1305,15 +1305,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1345,14 +1345,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.5, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.11", + "total_run_cost": "5.77", + "average_steps": "11.4", "percent_finished": "1.0" } }, @@ -1360,8 +1360,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1373,15 +1373,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1413,23 +1413,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "26.27", - "average_steps": "11.08", - "percent_finished": "1.0" + "average_agent_cost": "0.11", + "total_run_cost": "11.54", + "average_steps": "9.55", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1441,15 +1441,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1462,42 +1462,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.6, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.12", - "total_run_cost": "12.63", - "average_steps": "9.92", - "percent_finished": "0.98" + "average_agent_cost": "0.29", + "total_run_cost": "15.28", + "average_steps": "10.68", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1509,15 +1509,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1549,14 +1549,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", + "average_agent_cost": "0.25", + "total_run_cost": "26.27", + "average_steps": "11.08", "percent_finished": "1.0" } }, @@ -1564,8 +1564,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1577,15 +1577,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1598,33 +1598,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5253, + "score": 0.73, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.45", - "total_run_cost": "44.58", - "average_steps": "19.98", + "average_agent_cost": "0.11", + "total_run_cost": "12.27", + "average_steps": "10.33", "percent_finished": "1.0" } }, @@ -1632,8 +1632,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1645,15 +1645,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1666,34 +1666,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_telecom", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/telecom", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/telecom", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.11", + "total_run_cost": "12.27", + "average_steps": "10.33", + "percent_finished": "1.0" } }, "generation_config": { @@ -1721,7 +1721,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1753,23 +1753,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "11.54", - "average_steps": "9.55", - "percent_finished": "0.99" + "average_agent_cost": "0.12", + "total_run_cost": "12.63", + "average_steps": "9.92", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1781,15 +1781,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1821,23 +1821,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.1", - "total_run_cost": "15.15", - "average_steps": "9.36", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1849,8 +1849,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1925,7 +1925,7 @@ } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1957,23 +1957,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.53, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.15", - "total_run_cost": "18.88", - "average_steps": "9.92", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1985,15 +1985,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2006,33 +2006,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_telecom", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/telecom", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/telecom", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (telecom subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.53, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", + "average_agent_cost": "0.15", + "total_run_cost": "18.88", + "average_steps": "9.92", "percent_finished": "1.0" } }, @@ -2040,8 +2040,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2053,15 +2053,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2093,23 +2093,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.1", + "total_run_cost": "15.15", + "average_steps": "9.36", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2121,8 +2121,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } diff --git a/data/models/openai_gpt-5.2.json b/data/models/openai_gpt-5.2.json index 264a902cf7567e7522c491bb5946e45236c20f82..e0e38b46e0d5597f0e46a23c8cec09536809173d 100644 --- a/data/models/openai_gpt-5.2.json +++ b/data/models/openai_gpt-5.2.json @@ -10,7 +10,7 @@ }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.0, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.9, + "score": 54.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, + "score": 64.9, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.3-codex.json b/data/models/openai_gpt-5.3-codex.json index c2e0e3c5c76650f4f61bd284a560a52ddba90f4a..51d6b09ce0b1d962aca263f43e44500e909ca5b9 100644 --- a/data/models/openai_gpt-5.3-codex.json +++ b/data/models/openai_gpt-5.3-codex.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.6, + "score": 75.1, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 64.7, + "score": 74.6, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 75.1, + "score": 64.7, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.json b/data/models/openai_gpt-5.json index 560e42fd58c2bc4d9dd832525000e6de22d241c0..7f9b921e3b907893c3c276b576de2e26766ef3ff 100644 --- a/data/models/openai_gpt-5.json +++ b/data/models/openai_gpt-5.json @@ -10,7 +10,7 @@ }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 33.9, + "score": 49.6, "uncertainty": { "standard_error": { "value": 2.9 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 49.6, + "score": 33.9, "uncertainty": { "standard_error": { "value": 2.9 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-oss-120b.json b/data/models/openai_gpt-oss-120b.json index 28749d5a8860c22138e6ef73bbbf2d8523253aca..24fd3c5da7f833c16b44793fae74161552f381de 100644 --- a/data/models/openai_gpt-oss-120b.json +++ b/data/models/openai_gpt-oss-120b.json @@ -4,8 +4,8 @@ "id": "openai/gpt-oss-120b", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ @@ -313,7 +313,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-120b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-120b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -337,7 +337,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -346,17 +346,17 @@ "max_score": 100.0 }, "score_details": { - "score": 14.2, + "score": 18.7, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -373,7 +373,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -387,7 +387,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-120b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-120b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -411,7 +411,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -420,17 +420,17 @@ "max_score": 100.0 }, "score_details": { - "score": 18.7, + "score": 14.2, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -447,7 +447,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_o3-mini-2025-01-31.json b/data/models/openai_o3-mini-2025-01-31.json index ced88fa4c1ef868f5e43dfa70677a0ba0a9bbf22..02714848d39229c376c0dd42c9bbb1fc6ef4337f 100644 --- a/data/models/openai_o3-mini-2025-01-31.json +++ b/data/models/openai_o3-mini-2025-01-31.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/openai_o3-mini-2025-01-31/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/openai_o3-mini-2025-01-31/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/openai_o3-mini-2025-01-31/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/openai_o3-mini-2025-01-31/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/openai_o4-mini-2025-04-16.json b/data/models/openai_o4-mini-2025-04-16.json index 6d7680cf32b045c46e6d407e1e671e786eaccdaf..89affac90b976d3ab841c13f750b966ab7f3be0b 100644 --- a/data/models/openai_o4-mini-2025-04-16.json +++ b/data/models/openai_o4-mini-2025-04-16.json @@ -1,9 +1,9 @@ { "model_info": { "name": "o4-mini-2025-04-16", - "id": "openai/o4-mini-2025-04-16", - "developer": "openai", - "inference_platform": "openai" + "developer": "OpenAI", + "inference_platform": "openai", + "id": "openai/o4-mini-2025-04-16" }, "evaluations": [ { @@ -746,13 +746,13 @@ } }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", - "retrieved_timestamp": "1760492095.8105888", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", + "retrieved_timestamp": "1770683238.099205", "source_metadata": { - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation" + "source_type": "documentation", + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party" }, "eval_library": { "name": "unknown", @@ -762,62 +762,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", - "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.014084507042253521 - }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Medium Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Hard Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.30985915492957744 - }, + "score": 0.0143 + } + }, + { + "evaluation_name": "Medium Problems", "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Easy Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.8873239436619719 - }, + "score": 0.2923 + } + }, + { + "evaluation_name": "Easy Problems", "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] + }, + "metric_config": { + "evaluation_description": "Pass@1 on Easy Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.8571 } } ], @@ -825,13 +825,13 @@ "generation_config": null }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", - "retrieved_timestamp": "1770683238.099205", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", + "retrieved_timestamp": "1760492095.8105888", "source_metadata": { + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation", - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party" + "source_type": "documentation" }, "eval_library": { "name": "unknown", @@ -841,62 +841,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", + "metric_config": { + "evaluation_description": "Pass@1 on Hard Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.014084507042253521 + }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Medium Problems", "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.0143 - } - }, - { - "evaluation_name": "Medium Problems", + "score": 0.30985915492957744 + }, "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Easy Problems", "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Easy Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.2923 - } - }, - { - "evaluation_name": "Easy Problems", + "score": 0.8873239436619719 + }, "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] - }, - "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.8571 } } ], diff --git a/data/models/openbmb_Eurus-RM-7b.json b/data/models/openbmb_Eurus-RM-7b.json index 44637ca17276f082725f61866dcfa3229cc54274..e1154a660c89f219cec7ec844602d9d8fdfa08ad 100644 --- a/data/models/openbmb_Eurus-RM-7b.json +++ b/data/models/openbmb_Eurus-RM-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8159 + "score": 0.5806 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9804 + "score": 0.6 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5683 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8135 + "score": 0.6267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8633 + "score": 0.7475 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7172 + "score": 0.5972 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/openbmb_Eurus-RM-7b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_Eurus-RM-7b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5806 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6 + "score": 0.8159 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9804 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5683 + "score": 0.6557 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6267 + "score": 0.8135 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7475 + "score": 0.8633 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5972 + "score": 0.7172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openbmb_UltraRM-13b.json b/data/models/openbmb_UltraRM-13b.json index c52a509adb327ccd9798d5a844c78601940ebb17..84bdd483e26f91976f0250093c6ea14b4f0ff97c 100644 --- a/data/models/openbmb_UltraRM-13b.json +++ b/data/models/openbmb_UltraRM-13b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4683 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5063 + "score": 0.6903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5519 + "score": 0.5548 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5089 + "score": 0.5986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6081 + "score": 0.6244 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.7294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6903 + "score": 0.4683 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.5063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5548 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5519 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5986 + "score": 0.5089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6244 + "score": 0.6081 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7294 + "score": 0.3036 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/qingy2019_Oracle-14B.json b/data/models/qingy2019_Oracle-14B.json index 8507f87701e6c5d9f2fe6ae190596692bd22c9b5..a9e8d67b51aef6bfb306c8b0708d2a2d905b01ac 100644 --- a/data/models/qingy2019_Oracle-14B.json +++ b/data/models/qingy2019_Oracle-14B.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MixtralForCausalLM", "params_billions": "13.668" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2358 + "score": 0.2401 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4612 + "score": 0.4622 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0642 + "score": 0.0725 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2609 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3717 + "score": 0.3703 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2382 + "score": 0.2379 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2401 + "score": 0.2358 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4622 + "score": 0.4612 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0725 + "score": 0.0642 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2576 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3703 + "score": 0.3717 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2379 + "score": 0.2382 } } ], diff --git a/data/models/qingy2019_Qwen2.5-Math-14B-Instruct.json b/data/models/qingy2019_Qwen2.5-Math-14B-Instruct.json index d1f9d8445e75d7ec13d0b96a4671b887b0187f18..ff2f7e139c8bf093d2ceb119549e49bddb25ef31 100644 --- a/data/models/qingy2019_Qwen2.5-Math-14B-Instruct.json +++ b/data/models/qingy2019_Qwen2.5-Math-14B-Instruct.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.6005 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.635 + "score": 0.6356 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.2764 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3725 + "score": 0.3691 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5331 + "score": 0.5339 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6005 + "score": 0.6066 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6356 + "score": 0.635 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2764 + "score": 0.3716 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3691 + "score": 0.3725 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5339 + "score": 0.5331 } } ], diff --git a/data/models/recoilme_Gemma-2-Ataraxy-Gemmasutra-9B-slerp.json b/data/models/recoilme_Gemma-2-Ataraxy-Gemmasutra-9B-slerp.json index c05da714366e29f7f79e0eed58d76a1be751fa7a..43ba454093c078e0583e3f1326acb815bed70526 100644 --- a/data/models/recoilme_Gemma-2-Ataraxy-Gemmasutra-9B-slerp.json +++ b/data/models/recoilme_Gemma-2-Ataraxy-Gemmasutra-9B-slerp.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7649 + "score": 0.2854 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5974 + "score": 0.5984 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0174 + "score": 0.1005 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3305 + "score": 0.3297 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4245 + "score": 0.4607 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4207 + "score": 0.4162 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2854 + "score": 0.7649 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5984 + "score": 0.5974 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1005 + "score": 0.0174 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3297 + "score": 0.3305 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4607 + "score": 0.4245 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.4207 } } ], diff --git a/data/models/rombodawg_Rombos-LLM-V2.5.1-Qwen-3b.json b/data/models/rombodawg_Rombos-LLM-V2.5.1-Qwen-3b.json index 1fee301e22697604ff280086533e166dc0ffc3b8..d4e71fb3a6a3a9e4e33eca0e3f27e30893b0529f 100644 --- a/data/models/rombodawg_Rombos-LLM-V2.5.1-Qwen-3b.json +++ b/data/models/rombodawg_Rombos-LLM-V2.5.1-Qwen-3b.json @@ -5,7 +5,7 @@ "developer": "rombodawg", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.397" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2595 + "score": 0.2566 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3884 + "score": 0.39 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0914 + "score": 0.1208 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.2626 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2719 + "score": 0.2741 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2566 + "score": 0.2595 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.39 + "score": 0.3884 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1208 + "score": 0.0914 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2626 + "score": 0.2743 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2741 + "score": 0.2719 } } ], diff --git a/data/models/spow12_ChatWaifu_v2.0_22B.json b/data/models/spow12_ChatWaifu_v2.0_22B.json index 1a3badb0cb1336d664d2bb7c663a51b7fe206d12..3d3a0fa59a44de9bc7e3fe4a40205e4db6c78a5a 100644 --- a/data/models/spow12_ChatWaifu_v2.0_22B.json +++ b/data/models/spow12_ChatWaifu_v2.0_22B.json @@ -5,7 +5,7 @@ "developer": "spow12", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "22.247" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6517 + "score": 0.6511 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5908 + "score": 0.5926 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2032 + "score": 0.1858 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3247 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3812 + "score": 0.3836 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6511 + "score": 0.6517 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5926 + "score": 0.5908 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1858 + "score": 0.2032 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3247 + "score": 0.3238 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3836 + "score": 0.3812 } } ], diff --git a/data/models/utter-project_EuroLLM-9B.json b/data/models/utter-project_EuroLLM-9B.json new file mode 100644 index 0000000000000000000000000000000000000000..0c29dc58cf38083c6f8a63ab8be09d489f94ca96 --- /dev/null +++ b/data/models/utter-project_EuroLLM-9B.json @@ -0,0 +1,49 @@ +{ + "model_info": { + "name": "EuroLLM 9B", + "id": "utter-project/EuroLLM-9B", + "developer": "unknown" + }, + "evaluations": [ + { + "evaluation_id": "la_leaderboard/utter-project/EuroLLM-9B/1774451270", + "retrieved_timestamp": "2024-10-27T00:00:00Z", + "source_metadata": { + "source_name": "La Leaderboard", + "source_type": "evaluation_run", + "source_url": "https://huggingface.co/spaces/la-leaderboard/la-leaderboard", + "source_organization_name": "La Leaderboard", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "custom", + "version": "1.0" + }, + "benchmark": "la_leaderboard", + "evaluation_results": [ + { + "evaluation_name": "la_leaderboard", + "metric_config": { + "evaluation_description": "La Leaderboard: LLM evaluation for Spanish varieties and languages of Spain and Latin America", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 100 + }, + "score_details": { + "score": 25.87 + }, + "source_data": { + "source_type": "url", + "dataset_name": "La Leaderboard composite dataset", + "url": [ + "https://huggingface.co/spaces/la-leaderboard/la-leaderboard" + ] + } + } + ], + "detailed_evaluation_results": null, + "generation_config": null + } + ] +} \ No newline at end of file diff --git a/data/models/weqweasdas_RM-Mistral-7B.json b/data/models/weqweasdas_RM-Mistral-7B.json index 2c0c95b4657b4530753b94c6c05b68b220f49072..014b8589e308fa2571a3ca85971364da616341ae 100644 --- a/data/models/weqweasdas_RM-Mistral-7B.json +++ b/data/models/weqweasdas_RM-Mistral-7B.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.596 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5937 + "score": 0.7982 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5956 + "score": 0.6053 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6911 + "score": 0.8703 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.7736 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6226 + "score": 0.753 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7982 + "score": 0.596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.5937 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6053 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5956 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8703 + "score": 0.6911 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7736 + "score": 0.7293 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.753 + "score": 0.6226 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/xai_grok-4-0709.json b/data/models/xai_grok-4-0709.json index 17df1c3c71c8bde63dde1258e2fa6f3d90db167c..838e8a4c652b167f0ca9e369c9a16e0d914d31c1 100644 --- a/data/models/xai_grok-4-0709.json +++ b/data/models/xai_grok-4-0709.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/xai_grok-4-0709/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/xai_grok-4.json b/data/models/xai_grok-4.json index cecbb7f7fed16e61050c3064764dc9d378c04607..f874bd900d3ef43e4eee8abe6e14222905fb7f59 100644 --- a/data/models/xai_grok-4.json +++ b/data/models/xai_grok-4.json @@ -4,13 +4,13 @@ "id": "xai/grok-4", "developer": "xAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.4, + "score": 27.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 23.1, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-4/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-4/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 23.1, + "score": 25.4, "uncertainty": { "standard_error": { "value": 2.9 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok 4\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok 4\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/xai_grok-code-fast-1.json b/data/models/xai_grok-code-fast-1.json index ff7837aff9b5d5c58d5f68ff0cba6d9e84315c2e..3953141509cadc33315c3a6d77b9eef13f18557f 100644 --- a/data/models/xai_grok-code-fast-1.json +++ b/data/models/xai_grok-code-fast-1.json @@ -4,13 +4,13 @@ "id": "xai/grok-code-fast-1", "developer": "xAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 25.8, + "score": 14.2, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__grok-code-fast-1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__grok-code-fast-1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 14.2, + "score": 25.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Grok Code Fast 1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Grok Code Fast 1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/yam-peleg_Hebrew-Mistral-7B-200K.json b/data/models/yam-peleg_Hebrew-Mistral-7B-200K.json index 45e3ebc7caff20d8f6859d7b181b75e707f19ee6..cbb3bd227faa6622c732549ea7bd0545f1455ef0 100644 --- a/data/models/yam-peleg_Hebrew-Mistral-7B-200K.json +++ b/data/models/yam-peleg_Hebrew-Mistral-7B-200K.json @@ -5,7 +5,7 @@ "developer": "yam-peleg", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", "params_billions": "7.504" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.177 + "score": 0.1856 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3411 + "score": 0.4149 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.0234 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.276 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.374 + "score": 0.3765 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2529 + "score": 0.2573 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1856 + "score": 0.177 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.3411 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0234 + "score": 0.031 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3765 + "score": 0.374 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2573 + "score": 0.2529 } } ],