diff --git a/data/benchmarks.json b/data/benchmarks.json index b89fad043205746a1fe3c3d97fb375c96123e519..9571727dfcd0f5a5c4445684a0e339ea8b9100b5 100644 --- a/data/benchmarks.json +++ b/data/benchmarks.json @@ -57,7 +57,7 @@ }, { "benchmark": "reward-bench", - "model_count": 328 + "model_count": 327 }, { "benchmark": "swe-bench", diff --git a/data/benchmarks/appworld_test_normal.json b/data/benchmarks/appworld_test_normal.json index 27bfa4d8d1ae3df861f951b1a84e2236e37e45c3..df9e46daf5dd19a85e092a1af20da57930852dac 100644 --- a/data/benchmarks/appworld_test_normal.json +++ b/data/benchmarks/appworld_test_normal.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "appworld/test_normal": 0.7 + "appworld/test_normal": 0.64 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "appworld/test_normal": 0.55 + "appworld/test_normal": 0.36 } }, { diff --git a/data/benchmarks/browsecompplus.json b/data/benchmarks/browsecompplus.json index fa5cad4bfa9a7ba3be365ad38a5955a6ca39cc9b..34f802eb19d92b7a54aef0531fdda5bcd88b81cf 100644 --- a/data/benchmarks/browsecompplus.json +++ b/data/benchmarks/browsecompplus.json @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "browsecompplus": 0.3333 + "browsecompplus": 0.48 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "browsecompplus": 0.43 + "browsecompplus": 0.26 } } ] diff --git a/data/benchmarks/hfopenllm_v2.json b/data/benchmarks/hfopenllm_v2.json index caf5db92e1eb1042b92e8b5d4090c0a48c4f57e2..69cfd9d7907951c5385529f14c35f4b9a0820d64 100644 --- a/data/benchmarks/hfopenllm_v2.json +++ b/data/benchmarks/hfopenllm_v2.json @@ -1747,12 +1747,12 @@ "name": "NeuralDaredevil-SuperNova-Lite-7B-DARETIES-abliterated", "developer": "BoltMonkey", "scores": { - "IFEval": 0.459, - "BBH": 0.5185, - "MATH Level 5": 0.0937, - "GPQA": 0.2743, - "MUSR": 0.4083, - "MMLU-PRO": 0.3631 + "IFEval": 0.7999, + "BBH": 0.5152, + "MATH Level 5": 0.1193, + "GPQA": 0.281, + "MUSR": 0.4019, + "MMLU-PRO": 0.3733 } }, { @@ -2176,12 +2176,12 @@ "name": "LION-Gemma-2b-dpo-v1.0", "developer": "Columbia-NLP", "scores": { - "IFEval": 0.3102, - "BBH": 0.3881, - "MATH Level 5": 0.0536, - "GPQA": 0.2534, - "MUSR": 0.4081, - "MMLU-PRO": 0.1665 + "IFEval": 0.3278, + "BBH": 0.392, + "MATH Level 5": 0.0431, + "GPQA": 0.2492, + "MUSR": 0.412, + "MMLU-PRO": 0.1666 } }, { @@ -3229,12 +3229,12 @@ "name": "PathfinderAI", "developer": "Daemontatox", "scores": { - "IFEval": 0.4855, - "BBH": 0.6627, - "MATH Level 5": 0.4841, - "GPQA": 0.3096, - "MUSR": 0.4256, - "MMLU-PRO": 0.5542 + "IFEval": 0.3745, + "BBH": 0.6668, + "MATH Level 5": 0.4758, + "GPQA": 0.3943, + "MUSR": 0.4858, + "MMLU-PRO": 0.5593 } }, { @@ -4009,12 +4009,12 @@ "name": "Llama-3.2-1B-SPIN-iter0", "developer": "DavieLion", "scores": { - "IFEval": 0.1507, - "BBH": 0.293, - "MATH Level 5": 0.0, - "GPQA": 0.2534, + "IFEval": 0.1549, + "BBH": 0.2937, + "MATH Level 5": 0.006, + "GPQA": 0.2576, "MUSR": 0.3565, - "MMLU-PRO": 0.1125 + "MMLU-PRO": 0.1128 } }, { @@ -4048,12 +4048,12 @@ "name": "Llama-3.2-1B-SPIN-iter3", "developer": "DavieLion", "scores": { - "IFEval": 0.1336, - "BBH": 0.2975, - "MATH Level 5": 0.0068, - "GPQA": 0.2534, - "MUSR": 0.35, - "MMLU-PRO": 0.1128 + "IFEval": 0.1324, + "BBH": 0.2972, + "MATH Level 5": 0.0, + "GPQA": 0.2643, + "MUSR": 0.3527, + "MMLU-PRO": 0.1129 } }, { @@ -4321,12 +4321,12 @@ "name": "Llama-3.1-8b-ITA", "developer": "DeepMount00", "scores": { - "IFEval": 0.5365, - "BBH": 0.517, - "MATH Level 5": 0.1707, - "GPQA": 0.3062, - "MUSR": 0.4487, - "MMLU-PRO": 0.396 + "IFEval": 0.7917, + "BBH": 0.5109, + "MATH Level 5": 0.1088, + "GPQA": 0.2878, + "MUSR": 0.4136, + "MMLU-PRO": 0.3876 } }, { @@ -7025,12 +7025,12 @@ "name": "Fireball-Meta-Llama-3.1-8B-Instruct-Agent-0.003-128K-code-ds-auto", "developer": "EpistemeAI", "scores": { - "IFEval": 0.7305, - "BBH": 0.4649, - "MATH Level 5": 0.1397, - "GPQA": 0.2659, - "MUSR": 0.3209, - "MMLU-PRO": 0.348 + "IFEval": 0.7207, + "BBH": 0.461, + "MATH Level 5": 0.1314, + "GPQA": 0.2701, + "MUSR": 0.3432, + "MMLU-PRO": 0.3354 } }, { @@ -7675,12 +7675,12 @@ "name": "Herplete-LLM-Llama-3.1-8b", "developer": "Etherll", "scores": { - "IFEval": 0.6106, - "BBH": 0.5347, - "MATH Level 5": 0.1548, - "GPQA": 0.3146, - "MUSR": 0.3991, - "MMLU-PRO": 0.3752 + "IFEval": 0.4672, + "BBH": 0.5013, + "MATH Level 5": 0.0279, + "GPQA": 0.2861, + "MUSR": 0.386, + "MMLU-PRO": 0.3482 } }, { @@ -8455,12 +8455,12 @@ "name": "Josiefied-Qwen2.5-0.5B-Instruct-abliterated-v1", "developer": "Goekdeniz-Guelmez", "scores": { - "IFEval": 0.3472, - "BBH": 0.3268, - "MATH Level 5": 0.0891, - "GPQA": 0.2517, - "MUSR": 0.3262, - "MMLU-PRO": 0.1641 + "IFEval": 0.3417, + "BBH": 0.3292, + "MATH Level 5": 0.0023, + "GPQA": 0.2576, + "MUSR": 0.3249, + "MMLU-PRO": 0.1638 } }, { @@ -8702,12 +8702,12 @@ "name": "Nature-Reason-1.2-reallysmall", "developer": "GuilhermeNaturaUmana", "scores": { - "IFEval": 0.4985, - "BBH": 0.5645, - "MATH Level 5": 0.2576, - "GPQA": 0.3003, - "MUSR": 0.4373, - "MMLU-PRO": 0.4429 + "IFEval": 0.4791, + "BBH": 0.5649, + "MATH Level 5": 0.25, + "GPQA": 0.2995, + "MUSR": 0.4439, + "MMLU-PRO": 0.4408 } }, { @@ -9170,12 +9170,12 @@ "name": "SmolLM2-360M-Instruct", "developer": "HuggingFaceTB", "scores": { - "IFEval": 0.083, - "BBH": 0.3053, - "MATH Level 5": 0.0083, - "GPQA": 0.2651, - "MUSR": 0.3423, - "MMLU-PRO": 0.1126 + "IFEval": 0.3842, + "BBH": 0.3144, + "MATH Level 5": 0.0151, + "GPQA": 0.255, + "MUSR": 0.3461, + "MMLU-PRO": 0.1117 } }, { @@ -13031,12 +13031,12 @@ "name": "SpydazWeb_AI_HumanAI_012_INSTRUCT_IA", "developer": "LeroyDyer", "scores": { - "IFEval": 0.3036, - "BBH": 0.4575, + "IFEval": 0.3066, + "BBH": 0.4577, "MATH Level 5": 0.0446, - "GPQA": 0.3012, - "MUSR": 0.4253, - "MMLU-PRO": 0.2329 + "GPQA": 0.2995, + "MUSR": 0.4254, + "MMLU-PRO": 0.2318 } }, { @@ -13057,12 +13057,12 @@ "name": "SpydazWeb_AI_HumanAI_012_INSTRUCT_XA", "developer": "LeroyDyer", "scores": { - "IFEval": 0.3798, - "BBH": 0.4483, - "MATH Level 5": 0.04, - "GPQA": 0.3129, - "MUSR": 0.4148, - "MMLU-PRO": 0.2389 + "IFEval": 0.3579, + "BBH": 0.4477, + "MATH Level 5": 0.0423, + "GPQA": 0.3096, + "MUSR": 0.4134, + "MMLU-PRO": 0.2376 } }, { @@ -14305,12 +14305,12 @@ "name": "Llama-3-8B-Magpie-Align-v0.1", "developer": "Magpie-Align", "scores": { - "IFEval": 0.4027, - "BBH": 0.4789, - "MATH Level 5": 0.0461, - "GPQA": 0.2768, - "MUSR": 0.3087, - "MMLU-PRO": 0.3001 + "IFEval": 0.4118, + "BBH": 0.4811, + "MATH Level 5": 0.034, + "GPQA": 0.2752, + "MUSR": 0.3047, + "MMLU-PRO": 0.3006 } }, { @@ -16874,6 +16874,19 @@ "MMLU-PRO": 0.232 } }, + { + "model_id": "NousResearch/Yarn-Llama-2-7b-128k", + "name": "Yarn-Llama-2-7b-128k", + "developer": "NousResearch", + "scores": { + "IFEval": 0.1485, + "BBH": 0.3248, + "MATH Level 5": 0.0151, + "GPQA": 0.2601, + "MUSR": 0.3967, + "MMLU-PRO": 0.1791 + } + }, { "model_id": "NousResearch/Yarn-Llama-2-7b-64k", "name": "Yarn-Llama-2-7b-64k", @@ -18128,11 +18141,11 @@ "developer": "PrimeIntellect", "scores": { "IFEval": 0.1757, - "BBH": 0.274, + "BBH": 0.276, "MATH Level 5": 0.0, - "GPQA": 0.25, - "MUSR": 0.3753, - "MMLU-PRO": 0.112 + "GPQA": 0.2534, + "MUSR": 0.3339, + "MMLU-PRO": 0.1123 } }, { @@ -18257,12 +18270,12 @@ "name": "Casa-14b-sce", "developer": "Quazim0t0", "scores": { - "IFEval": 0.6654, - "BBH": 0.6901, - "MATH Level 5": 0.4698, - "GPQA": 0.3331, - "MUSR": 0.431, - "MMLU-PRO": 0.5426 + "IFEval": 0.6718, + "BBH": 0.6891, + "MATH Level 5": 0.4985, + "GPQA": 0.3339, + "MUSR": 0.4323, + "MMLU-PRO": 0.5408 } }, { @@ -18699,12 +18712,12 @@ "name": "ODB-14B-sce", "developer": "Quazim0t0", "scores": { - "IFEval": 0.2922, - "BBH": 0.6559, - "MATH Level 5": 0.2545, - "GPQA": 0.2659, - "MUSR": 0.3929, - "MMLU-PRO": 0.5207 + "IFEval": 0.7016, + "BBH": 0.6942, + "MATH Level 5": 0.4116, + "GPQA": 0.3624, + "MUSR": 0.4571, + "MMLU-PRO": 0.5411 } }, { @@ -19453,12 +19466,12 @@ "name": "Qwen2.5-0.5B-Instruct", "developer": "Qwen", "scores": { - "IFEval": 0.3071, - "BBH": 0.3341, - "MATH Level 5": 0.0, - "GPQA": 0.2576, - "MUSR": 0.3329, - "MMLU-PRO": 0.1697 + "IFEval": 0.3153, + "BBH": 0.3322, + "MATH Level 5": 0.1035, + "GPQA": 0.2592, + "MUSR": 0.3342, + "MMLU-PRO": 0.172 } }, { @@ -19973,12 +19986,12 @@ "name": "Replete-LLM-Qwen2-7b", "developer": "Replete-AI", "scores": { - "IFEval": 0.0905, - "BBH": 0.2985, + "IFEval": 0.0932, + "BBH": 0.2977, "MATH Level 5": 0.0, - "GPQA": 0.2534, - "MUSR": 0.3848, - "MMLU-PRO": 0.1158 + "GPQA": 0.2475, + "MUSR": 0.3941, + "MMLU-PRO": 0.1157 } }, { @@ -25056,12 +25069,12 @@ "name": "Llama3.1-8B-Cobalt", "developer": "ValiantLabs", "scores": { - "IFEval": 0.7168, - "BBH": 0.4911, - "MATH Level 5": 0.1533, - "GPQA": 0.2861, - "MUSR": 0.3512, - "MMLU-PRO": 0.3663 + "IFEval": 0.3496, + "BBH": 0.4947, + "MATH Level 5": 0.1269, + "GPQA": 0.3037, + "MUSR": 0.3959, + "MMLU-PRO": 0.3644 } }, { @@ -26499,12 +26512,12 @@ "name": "autotrain-0tmgq-5tpbg", "developer": "abhishek", "scores": { - "IFEval": 0.1952, - "BBH": 0.3127, - "MATH Level 5": 0.0128, - "GPQA": 0.2592, - "MUSR": 0.3584, - "MMLU-PRO": 0.1144 + "IFEval": 0.1957, + "BBH": 0.3135, + "MATH Level 5": 0.0, + "GPQA": 0.2517, + "MUSR": 0.365, + "MMLU-PRO": 0.1151 } }, { @@ -26590,12 +26603,12 @@ "name": "QAIMath-Qwen2.5-7B-TIES", "developer": "adriszmar", "scores": { - "IFEval": 0.1746, - "BBH": 0.3126, - "MATH Level 5": 0.0, - "GPQA": 0.245, - "MUSR": 0.4096, - "MMLU-PRO": 0.1087 + "IFEval": 0.1685, + "BBH": 0.3124, + "MATH Level 5": 0.0015, + "GPQA": 0.2492, + "MUSR": 0.3963, + "MMLU-PRO": 0.1066 } }, { @@ -26876,12 +26889,12 @@ "name": "Llama-3.1-Storm-8B", "developer": "akjindal53244", "scores": { - "IFEval": 0.8033, - "BBH": 0.5196, - "MATH Level 5": 0.1624, - "GPQA": 0.3096, + "IFEval": 0.8051, + "BBH": 0.5189, + "MATH Level 5": 0.1722, + "GPQA": 0.3263, "MUSR": 0.4028, - "MMLU-PRO": 0.3812 + "MMLU-PRO": 0.3803 } }, { @@ -26902,12 +26915,12 @@ "name": "Llama-3.1-Tulu-3-70B", "developer": "allenai", "scores": { - "IFEval": 0.8291, - "BBH": 0.6164, - "MATH Level 5": 0.4502, + "IFEval": 0.8379, + "BBH": 0.6157, + "MATH Level 5": 0.3829, "GPQA": 0.3733, - "MUSR": 0.4948, - "MMLU-PRO": 0.4645 + "MUSR": 0.4988, + "MMLU-PRO": 0.4656 } }, { @@ -26941,12 +26954,12 @@ "name": "Llama-3.1-Tulu-3-8B", "developer": "allenai", "scores": { - "IFEval": 0.8267, - "BBH": 0.405, - "MATH Level 5": 0.1964, - "GPQA": 0.2987, + "IFEval": 0.8255, + "BBH": 0.4061, + "MATH Level 5": 0.2115, + "GPQA": 0.297, "MUSR": 0.4175, - "MMLU-PRO": 0.2827 + "MMLU-PRO": 0.2821 } }, { @@ -30347,12 +30360,12 @@ "name": "Llama-3.2-3B-Deep-Test", "developer": "bunnycore", "scores": { - "IFEval": 0.1775, - "BBH": 0.295, - "MATH Level 5": 0.0, - "GPQA": 0.2517, - "MUSR": 0.3647, - "MMLU-PRO": 0.1049 + "IFEval": 0.4652, + "BBH": 0.4531, + "MATH Level 5": 0.1284, + "GPQA": 0.2643, + "MUSR": 0.3394, + "MMLU-PRO": 0.3152 } }, { @@ -32154,12 +32167,12 @@ "name": "Llama-3-8B-Orpo-v0.1", "developer": "dfurman", "scores": { - "IFEval": 0.3, - "BBH": 0.3853, - "MATH Level 5": 0.0415, - "GPQA": 0.2617, - "MUSR": 0.3579, - "MMLU-PRO": 0.2281 + "IFEval": 0.2835, + "BBH": 0.3842, + "MATH Level 5": 0.0521, + "GPQA": 0.2609, + "MUSR": 0.3566, + "MMLU-PRO": 0.2298 } }, { @@ -37258,19 +37271,6 @@ "MMLU-PRO": 0.3103 } }, - { - "model_id": "icefog72/IceSakeV6RP-7b", - "name": "IceSakeV6RP-7b", - "developer": "icefog72", - "scores": { - "IFEval": 0.5033, - "BBH": 0.4976, - "MATH Level 5": 0.0619, - "GPQA": 0.2911, - "MUSR": 0.42, - "MMLU-PRO": 0.3093 - } - }, { "model_id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", @@ -37692,12 +37692,12 @@ "name": "Kosmos-EVAA-Fusion-8B", "developer": "jaspionjader", "scores": { - "IFEval": 0.4418, - "BBH": 0.5406, - "MATH Level 5": 0.1352, - "GPQA": 0.3062, + "IFEval": 0.4345, + "BBH": 0.5419, + "MATH Level 5": 0.1292, + "GPQA": 0.3087, "MUSR": 0.4277, - "MMLU-PRO": 0.386 + "MMLU-PRO": 0.3854 } }, { @@ -43971,12 +43971,12 @@ "name": "Phi-3-mini-4k-instruct", "developer": "microsoft", "scores": { - "IFEval": 0.5477, - "BBH": 0.5491, - "MATH Level 5": 0.1639, - "GPQA": 0.3322, - "MUSR": 0.4284, - "MMLU-PRO": 0.4022 + "IFEval": 0.5613, + "BBH": 0.5676, + "MATH Level 5": 0.1163, + "GPQA": 0.3196, + "MUSR": 0.395, + "MMLU-PRO": 0.3866 } }, { @@ -44101,12 +44101,12 @@ "name": "phi-4", "developer": "microsoft", "scores": { - "IFEval": 0.0585, - "BBH": 0.6691, - "MATH Level 5": 0.3165, - "GPQA": 0.406, + "IFEval": 0.0488, + "BBH": 0.6703, + "MATH Level 5": 0.2787, + "GPQA": 0.401, "MUSR": 0.5034, - "MMLU-PRO": 0.5287 + "MMLU-PRO": 0.5295 } }, { @@ -44413,12 +44413,12 @@ "name": "Mistral-Small-Instruct-2409", "developer": "mistralai", "scores": { - "IFEval": 0.667, - "BBH": 0.5213, - "MATH Level 5": 0.1435, - "GPQA": 0.3238, - "MUSR": 0.3632, - "MMLU-PRO": 0.396 + "IFEval": 0.6283, + "BBH": 0.583, + "MATH Level 5": 0.2039, + "GPQA": 0.3331, + "MUSR": 0.4063, + "MMLU-PRO": 0.4099 } }, { @@ -44465,12 +44465,12 @@ "name": "Mixtral-8x7B-v0.1", "developer": "mistralai", "scores": { - "IFEval": 0.2415, - "BBH": 0.5087, - "MATH Level 5": 0.102, - "GPQA": 0.3138, - "MUSR": 0.4321, - "MMLU-PRO": 0.385 + "IFEval": 0.2326, + "BBH": 0.5098, + "MATH Level 5": 0.0937, + "GPQA": 0.3205, + "MUSR": 0.4413, + "MMLU-PRO": 0.3871 } }, { @@ -44725,12 +44725,12 @@ "name": "NeuralDaredevil-8B-abliterated", "developer": "mlabonne", "scores": { - "IFEval": 0.7561, - "BBH": 0.5111, - "MATH Level 5": 0.0906, - "GPQA": 0.3062, - "MUSR": 0.4019, - "MMLU-PRO": 0.3841 + "IFEval": 0.4162, + "BBH": 0.5124, + "MATH Level 5": 0.0853, + "GPQA": 0.3029, + "MUSR": 0.415, + "MMLU-PRO": 0.3802 } }, { @@ -47598,12 +47598,12 @@ "name": "wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue", "developer": "ontocord", "scores": { - "IFEval": 0.1162, - "BBH": 0.3184, - "MATH Level 5": 0.0076, - "GPQA": 0.2634, - "MUSR": 0.3447, - "MMLU-PRO": 0.1124 + "IFEval": 0.1128, + "BBH": 0.3171, + "MATH Level 5": 0.0113, + "GPQA": 0.2685, + "MUSR": 0.346, + "MMLU-PRO": 0.1129 } }, { @@ -48716,12 +48716,12 @@ "name": "Llama-3-8B-ProLong-512k-Instruct", "developer": "princeton-nlp", "scores": { - "IFEval": 0.3978, - "BBH": 0.4983, - "MATH Level 5": 0.0582, - "GPQA": 0.281, - "MUSR": 0.425, - "MMLU-PRO": 0.3246 + "IFEval": 0.5508, + "BBH": 0.5028, + "MATH Level 5": 0.0529, + "GPQA": 0.2861, + "MUSR": 0.4266, + "MMLU-PRO": 0.3231 } }, { @@ -49431,12 +49431,12 @@ "name": "Calcium-Opus-14B-Elite", "developer": "prithivMLmods", "scores": { - "IFEval": 0.6052, - "BBH": 0.6317, - "MATH Level 5": 0.4789, - "GPQA": 0.3742, - "MUSR": 0.486, - "MMLU-PRO": 0.5302 + "IFEval": 0.6064, + "BBH": 0.6296, + "MATH Level 5": 0.3708, + "GPQA": 0.3733, + "MUSR": 0.4873, + "MMLU-PRO": 0.5307 } }, { @@ -50861,12 +50861,12 @@ "name": "Qwen2.5-Math-14B-Instruct", "developer": "qingy2019", "scores": { - "IFEval": 0.6005, - "BBH": 0.6356, - "MATH Level 5": 0.2764, - "GPQA": 0.3691, + "IFEval": 0.6066, + "BBH": 0.635, + "MATH Level 5": 0.3716, + "GPQA": 0.3725, "MUSR": 0.4757, - "MMLU-PRO": 0.5339 + "MMLU-PRO": 0.5331 } }, { @@ -51316,12 +51316,12 @@ "name": "recoilme-gemma-2-9B-v0.2", "developer": "recoilme", "scores": { - "IFEval": 0.7592, - "BBH": 0.6026, - "MATH Level 5": 0.0529, - "GPQA": 0.3289, - "MUSR": 0.4099, - "MMLU-PRO": 0.4163 + "IFEval": 0.2747, + "BBH": 0.6031, + "MATH Level 5": 0.0831, + "GPQA": 0.3305, + "MUSR": 0.4686, + "MMLU-PRO": 0.4122 } }, { @@ -51329,12 +51329,12 @@ "name": "recoilme-gemma-2-9B-v0.3", "developer": "recoilme", "scores": { - "IFEval": 0.5761, - "BBH": 0.602, - "MATH Level 5": 0.1888, - "GPQA": 0.3372, - "MUSR": 0.4632, - "MMLU-PRO": 0.4039 + "IFEval": 0.7439, + "BBH": 0.5993, + "MATH Level 5": 0.0876, + "GPQA": 0.3238, + "MUSR": 0.4204, + "MMLU-PRO": 0.4072 } }, { @@ -53188,12 +53188,12 @@ "name": "Qwenvergence-14B-v3-Reason", "developer": "sometimesanotion", "scores": { - "IFEval": 0.5367, - "BBH": 0.6561, - "MATH Level 5": 0.358, - "GPQA": 0.3867, - "MUSR": 0.474, - "MMLU-PRO": 0.5395 + "IFEval": 0.5278, + "BBH": 0.6557, + "MATH Level 5": 0.3119, + "GPQA": 0.3842, + "MUSR": 0.4754, + "MMLU-PRO": 0.5396 } }, { @@ -54332,12 +54332,12 @@ "name": "lambda-gemma-2-9b-dpo", "developer": "tanliboy", "scores": { - "IFEval": 0.4501, - "BBH": 0.5472, - "MATH Level 5": 0.0944, - "GPQA": 0.3138, - "MUSR": 0.4017, - "MMLU-PRO": 0.3792 + "IFEval": 0.1829, + "BBH": 0.5488, + "MATH Level 5": 0.0, + "GPQA": 0.3104, + "MUSR": 0.4056, + "MMLU-PRO": 0.3805 } }, { @@ -56932,12 +56932,12 @@ "name": "Hebrew-Mistral-7B-200K", "developer": "yam-peleg", "scores": { - "IFEval": 0.1856, - "BBH": 0.4149, - "MATH Level 5": 0.0234, - "GPQA": 0.276, - "MUSR": 0.3765, - "MMLU-PRO": 0.2573 + "IFEval": 0.177, + "BBH": 0.3411, + "MATH Level 5": 0.031, + "GPQA": 0.2534, + "MUSR": 0.374, + "MMLU-PRO": 0.2529 } }, { diff --git a/data/benchmarks/livecodebenchpro.json b/data/benchmarks/livecodebenchpro.json index 896b3c77b4f3fcb0324b8662401bdd6aa70bb19b..f8f77204727bc54bca1bbaaa204ca44d2b265f12 100644 --- a/data/benchmarks/livecodebenchpro.json +++ b/data/benchmarks/livecodebenchpro.json @@ -205,9 +205,9 @@ "name": "gpt-5-2025-08-07", "developer": "OpenAI", "scores": { - "Hard Problems": 0.04225352112676056, - "Medium Problems": 0.4084507042253521, - "Easy Problems": 0.8873239436619719 + "Hard Problems": 0.0423, + "Medium Problems": 0.4085, + "Easy Problems": 0.9014 } }, { @@ -255,9 +255,9 @@ "name": "o4-mini-2025-04-16", "developer": "OpenAI", "scores": { - "Hard Problems": 0.014084507042253521, - "Medium Problems": 0.30985915492957744, - "Easy Problems": 0.8873239436619719 + "Hard Problems": 0.0143, + "Medium Problems": 0.2923, + "Easy Problems": 0.8571 } }, { diff --git a/data/benchmarks/reward-bench.json b/data/benchmarks/reward-bench.json index d41f3c4a1d71f52cfb866cd53b58580a0eae02dd..71717215bb28d07da86d76055db6b05b5dbcd648 100644 --- a/data/benchmarks/reward-bench.json +++ b/data/benchmarks/reward-bench.json @@ -81,17 +81,17 @@ "name": "CIR-AMS/BTRM_Qwen2_7b_0613", "developer": "CIR-AMS", "scores": { - "Score": 0.8172, + "Score": 0.5736, + "Chat": 0.9749, + "Chat Hard": 0.5724, + "Safety": 0.7178, + "Reasoning": 0.8775, + "Prior Sets (0.5 weight)": 0.7029, "Factuality": 0.5347, "Precise IF": 0.3563, "Math": 0.6066, - "Safety": 0.9014, "Focus": 0.5737, - "Ties": 0.6527, - "Chat": 0.9749, - "Chat Hard": 0.5724, - "Reasoning": 0.8775, - "Prior Sets (0.5 weight)": 0.7029 + "Ties": 0.6527 } }, { @@ -453,16 +453,16 @@ "name": "LxzGordon/URM-LLaMa-3.1-8B", "developer": "LxzGordon", "scores": { - "Score": 0.7394, - "Chat": 0.9553, - "Chat Hard": 0.8816, - "Safety": 0.9178, - "Reasoning": 0.9698, + "Score": 0.9294, "Factuality": 0.6884, "Precise IF": 0.45, "Math": 0.6393, + "Safety": 0.9108, "Focus": 0.9758, - "Ties": 0.7653 + "Ties": 0.7653, + "Chat": 0.9553, + "Chat Hard": 0.8816, + "Reasoning": 0.9698 } }, { @@ -555,17 +555,17 @@ "name": "OpenAssistant/oasst-rm-2-pythia-6.9b-epoch-1", "developer": "OpenAssistant", "scores": { - "Score": 0.2653, - "Chat": 0.9246, - "Chat Hard": 0.3728, - "Safety": 0.3289, - "Reasoning": 0.5855, - "Prior Sets (0.5 weight)": 0.6801, + "Score": 0.615, "Factuality": 0.3979, "Precise IF": 0.2875, "Math": 0.377, + "Safety": 0.5446, "Focus": 0.1535, - "Ties": 0.047 + "Ties": 0.047, + "Chat": 0.9246, + "Chat Hard": 0.3728, + "Reasoning": 0.5855, + "Prior Sets (0.5 weight)": 0.6801 } }, { @@ -609,17 +609,17 @@ "name": "PKU-Alignment/beaver-7b-v1.0-cost", "developer": "PKU-Alignment", "scores": { - "Score": 0.5798, + "Score": 0.3332, + "Chat": 0.6173, + "Chat Hard": 0.4232, + "Safety": 0.7589, + "Reasoning": 0.5482, + "Prior Sets (0.5 weight)": 0.57, "Factuality": 0.3263, "Precise IF": 0.2313, "Math": 0.3989, - "Safety": 0.7351, "Focus": 0.2939, - "Ties": -0.01, - "Chat": 0.6173, - "Chat Hard": 0.4232, - "Reasoning": 0.5482, - "Prior Sets (0.5 weight)": 0.57 + "Ties": -0.01 } }, { @@ -627,17 +627,17 @@ "name": "PKU-Alignment/beaver-7b-v1.0-reward", "developer": "PKU-Alignment", "scores": { - "Score": 0.1606, - "Chat": 0.8184, - "Chat Hard": 0.2873, - "Safety": 0.1422, - "Reasoning": 0.346, - "Prior Sets (0.5 weight)": 0.5993, + "Score": 0.4727, "Factuality": 0.2105, "Precise IF": 0.2938, "Math": 0.2623, + "Safety": 0.3757, "Focus": 0.0646, - "Ties": -0.01 + "Ties": -0.01, + "Chat": 0.8184, + "Chat Hard": 0.2873, + "Reasoning": 0.346, + "Prior Sets (0.5 weight)": 0.5993 } }, { @@ -663,17 +663,17 @@ "name": "PKU-Alignment/beaver-7b-v2.0-reward", "developer": "PKU-Alignment", "scores": { - "Score": 0.2544, - "Chat": 0.8994, - "Chat Hard": 0.364, - "Safety": 0.3156, - "Reasoning": 0.6887, - "Prior Sets (0.5 weight)": 0.6171, + "Score": 0.6366, "Factuality": 0.2168, "Precise IF": 0.2562, "Math": 0.3825, + "Safety": 0.6041, "Focus": 0.2606, - "Ties": 0.0944 + "Ties": 0.0944, + "Chat": 0.8994, + "Chat Hard": 0.364, + "Reasoning": 0.6887, + "Prior Sets (0.5 weight)": 0.6171 } }, { @@ -904,16 +904,16 @@ "name": "Ray2333/GRM-Llama3-8B-rewardmodel-ft", "developer": "Ray2333", "scores": { - "Score": 0.6766, - "Chat": 0.9553, - "Chat Hard": 0.8618, - "Safety": 0.9222, - "Reasoning": 0.9362, + "Score": 0.9154, "Factuality": 0.6274, "Precise IF": 0.35, "Math": 0.5847, + "Safety": 0.9081, "Focus": 0.8929, - "Ties": 0.6824 + "Ties": 0.6824, + "Chat": 0.9553, + "Chat Hard": 0.8618, + "Reasoning": 0.9362 } }, { @@ -921,16 +921,16 @@ "name": "Ray2333/GRM-gemma2-2B-rewardmodel-ft", "developer": "Ray2333", "scores": { - "Score": 0.5966, - "Chat": 0.9302, - "Chat Hard": 0.7719, - "Safety": 0.9222, - "Reasoning": 0.912, + "Score": 0.8839, "Factuality": 0.5305, "Precise IF": 0.3125, "Math": 0.5902, + "Safety": 0.9216, "Focus": 0.7455, - "Ties": 0.4788 + "Ties": 0.4788, + "Chat": 0.9302, + "Chat Hard": 0.7719, + "Reasoning": 0.912 } }, { @@ -956,17 +956,17 @@ "name": "Ray2333/GRM-llama3-8B-sftreg", "developer": "Ray2333", "scores": { - "Score": 0.8542, + "Score": 0.6089, + "Chat": 0.986, + "Chat Hard": 0.6776, + "Safety": 0.7867, + "Reasoning": 0.9229, + "Prior Sets (0.5 weight)": 0.7309, "Factuality": 0.6189, "Precise IF": 0.3875, "Math": 0.5792, - "Safety": 0.8919, "Focus": 0.6828, - "Ties": 0.5981, - "Chat": 0.986, - "Chat Hard": 0.6776, - "Reasoning": 0.9229, - "Prior Sets (0.5 weight)": 0.7309 + "Ties": 0.5981 } }, { @@ -1098,16 +1098,16 @@ "name": "ShikaiChen/LDL-Reward-Gemma-2-27B-v0.1", "developer": "ShikaiChen", "scores": { - "Score": 0.9499, + "Score": 0.7249, + "Chat": 0.9637, + "Chat Hard": 0.9079, + "Safety": 0.9222, + "Reasoning": 0.9903, "Factuality": 0.7558, "Precise IF": 0.35, "Math": 0.6448, - "Safety": 0.9378, "Focus": 0.9131, - "Ties": 0.7633, - "Chat": 0.9637, - "Chat Hard": 0.9079, - "Reasoning": 0.9903 + "Ties": 0.7633 } }, { @@ -1139,16 +1139,16 @@ "name": "Skywork/Skywork-Reward-Gemma-2-27B", "developer": "Skywork", "scores": { - "Score": 0.7576, - "Chat": 0.9581, - "Chat Hard": 0.9145, - "Safety": 0.9422, - "Reasoning": 0.9606, + "Score": 0.938, "Factuality": 0.7368, "Precise IF": 0.4031, "Math": 0.7049, + "Safety": 0.9189, "Focus": 0.9323, - "Ties": 0.8261 + "Ties": 0.8261, + "Chat": 0.9581, + "Chat Hard": 0.9145, + "Reasoning": 0.9606 } }, { @@ -1305,16 +1305,16 @@ "name": "Skywork/Skywork-VL-Reward-7B", "developer": "Skywork", "scores": { - "Score": 0.6885, - "Chat": 0.8994, - "Chat Hard": 0.875, - "Safety": 0.8911, - "Reasoning": 0.9176, + "Score": 0.9007, "Factuality": 0.6063, "Precise IF": 0.35, "Math": 0.6339, + "Safety": 0.9108, "Focus": 0.8909, - "Ties": 0.7586 + "Ties": 0.7586, + "Chat": 0.8994, + "Chat Hard": 0.875, + "Reasoning": 0.9176 } }, { @@ -1379,10 +1379,10 @@ "name": "ai2/tulu-2-7b-rm-v0-nectar-binarized-3.8m-check...", "developer": "AI2", "scores": { - "Score": 0.6924, - "Chat": 0.9441, - "Chat Hard": 0.3575, - "Safety": 0.7757 + "Score": 0.6895, + "Chat": 0.9385, + "Chat Hard": 0.3706, + "Safety": 0.7595 } }, { @@ -1441,17 +1441,17 @@ "name": "allenai/Llama-3.1-8B-Base-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.649, - "Chat": 0.933, - "Chat Hard": 0.7785, - "Safety": 0.8267, - "Reasoning": 0.7886, - "Prior Sets (0.5 weight)": 0.0, + "Score": 0.8463, "Factuality": 0.72, "Precise IF": 0.3625, "Math": 0.612, + "Safety": 0.8851, "Focus": 0.8323, - "Ties": 0.5406 + "Ties": 0.5406, + "Chat": 0.933, + "Chat Hard": 0.7785, + "Reasoning": 0.7886, + "Prior Sets (0.5 weight)": 0.0 } }, { @@ -1477,17 +1477,17 @@ "name": "allenai/Llama-3.1-Tulu-3-70B-SFT-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8892, + "Score": 0.722, + "Chat": 0.9693, + "Chat Hard": 0.8268, + "Safety": 0.8689, + "Reasoning": 0.8583, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.8084, "Precise IF": 0.3688, "Math": 0.6776, - "Safety": 0.9027, "Focus": 0.7778, - "Ties": 0.8308, - "Chat": 0.9693, - "Chat Hard": 0.8268, - "Reasoning": 0.8583, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.8308 } }, { @@ -1495,17 +1495,17 @@ "name": "allenai/Llama-3.1-Tulu-3-8B-DPO-RM-RB2", "developer": "allenai", "scores": { - "Score": 0.8431, + "Score": 0.687, + "Chat": 0.9553, + "Chat Hard": 0.761, + "Safety": 0.86, + "Reasoning": 0.7898, + "Prior Sets (0.5 weight)": 0.0, "Factuality": 0.7516, "Precise IF": 0.3875, "Math": 0.6284, - "Safety": 0.8662, "Focus": 0.8545, - "Ties": 0.6397, - "Chat": 0.9553, - "Chat Hard": 0.761, - "Reasoning": 0.7898, - "Prior Sets (0.5 weight)": 0.0 + "Ties": 0.6397 } }, { @@ -2085,20 +2085,6 @@ "Ties": 0.3534 } }, - { - "model_id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "developer": "allenai", - "scores": { - "Score": 0.5151, - "Factuality": 0.6484, - "Precise IF": 0.3312, - "Math": 0.5574, - "Safety": 0.7289, - "Focus": 0.4889, - "Ties": 0.3357 - } - }, { "model_id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", @@ -3486,17 +3472,17 @@ "name": "Claude 3 Haiku 20240307", "developer": "Anthropic", "scores": { - "Score": 0.3711, - "Chat": 0.9274, - "Chat Hard": 0.5197, - "Safety": 0.595, - "Reasoning": 0.706, - "Prior Sets (0.5 weight)": 0.6635, + "Score": 0.7289, "Factuality": 0.4042, "Precise IF": 0.2812, "Math": 0.3552, + "Safety": 0.7953, "Focus": 0.501, - "Ties": 0.0899 + "Ties": 0.0899, + "Chat": 0.9274, + "Chat Hard": 0.5197, + "Reasoning": 0.706, + "Prior Sets (0.5 weight)": 0.6635 } }, { @@ -3504,16 +3490,16 @@ "name": "Claude 3 Opus 20240229", "developer": "Anthropic", "scores": { - "Score": 0.5744, - "Chat": 0.9469, - "Chat Hard": 0.6031, - "Safety": 0.8378, - "Reasoning": 0.7868, + "Score": 0.8008, "Factuality": 0.5389, "Precise IF": 0.3312, "Math": 0.5137, + "Safety": 0.8662, "Focus": 0.6646, - "Ties": 0.5601 + "Ties": 0.5601, + "Chat": 0.9469, + "Chat Hard": 0.6031, + "Reasoning": 0.7868 } }, { @@ -3784,16 +3770,16 @@ "name": "infly/INF-ORM-Llama3.1-70B", "developer": "infly", "scores": { - "Score": 0.9511, + "Score": 0.7648, + "Chat": 0.9665, + "Chat Hard": 0.9101, + "Safety": 0.9644, + "Reasoning": 0.9912, "Factuality": 0.7411, "Precise IF": 0.4188, "Math": 0.6995, - "Safety": 0.9365, "Focus": 0.903, - "Ties": 0.8622, - "Chat": 0.9665, - "Chat Hard": 0.9101, - "Reasoning": 0.9912 + "Ties": 0.8622 } }, { @@ -3801,16 +3787,16 @@ "name": "internlm/internlm2-1_8b-reward", "developer": "internlm", "scores": { - "Score": 0.3902, - "Chat": 0.9358, - "Chat Hard": 0.6623, - "Safety": 0.4711, - "Reasoning": 0.8724, + "Score": 0.8217, "Factuality": 0.2758, "Precise IF": 0.3625, "Math": 0.4426, + "Safety": 0.8162, "Focus": 0.596, - "Ties": 0.1934 + "Ties": 0.1934, + "Chat": 0.9358, + "Chat Hard": 0.6623, + "Reasoning": 0.8724 } }, { @@ -3835,16 +3821,16 @@ "name": "internlm/internlm2-7b-reward", "developer": "internlm", "scores": { - "Score": 0.5335, - "Chat": 0.9916, - "Chat Hard": 0.6952, - "Safety": 0.5956, - "Reasoning": 0.9453, + "Score": 0.8759, "Factuality": 0.4211, "Precise IF": 0.4, "Math": 0.5628, + "Safety": 0.8716, "Focus": 0.7051, - "Ties": 0.5164 + "Ties": 0.5164, + "Chat": 0.9916, + "Chat Hard": 0.6952, + "Reasoning": 0.9453 } }, { @@ -4014,16 +4000,16 @@ "name": "nicolinho/QRM-Gemma-2-27B", "developer": "nicolinho", "scores": { - "Score": 0.7667, - "Chat": 0.9665, - "Chat Hard": 0.9013, - "Safety": 0.9578, - "Reasoning": 0.9826, + "Score": 0.9444, "Factuality": 0.7853, "Precise IF": 0.3719, "Math": 0.6995, + "Safety": 0.927, "Focus": 0.9535, - "Ties": 0.8321 + "Ties": 0.8321, + "Chat": 0.9665, + "Chat Hard": 0.9013, + "Reasoning": 0.9826 } }, { @@ -4055,16 +4041,16 @@ "name": "nicolinho/QRM-Llama3.1-8B-v2", "developer": "nicolinho", "scores": { - "Score": 0.7074, - "Chat": 0.9637, - "Chat Hard": 0.8684, - "Safety": 0.9467, - "Reasoning": 0.9677, + "Score": 0.9314, "Factuality": 0.6653, "Precise IF": 0.4062, "Math": 0.612, + "Safety": 0.9257, "Focus": 0.8909, - "Ties": 0.7234 + "Ties": 0.7234, + "Chat": 0.9637, + "Chat Hard": 0.8684, + "Reasoning": 0.9677 } }, { @@ -4202,16 +4188,16 @@ "name": "GPT-4o 2024-08-06", "developer": "OpenAI", "scores": { - "Score": 0.8673, + "Score": 0.6493, + "Chat": 0.9609, + "Chat Hard": 0.761, + "Safety": 0.8619, + "Reasoning": 0.8661, "Factuality": 0.5684, "Precise IF": 0.3312, "Math": 0.623, - "Safety": 0.8811, "Focus": 0.7293, - "Ties": 0.7819, - "Chat": 0.9609, - "Chat Hard": 0.761, - "Reasoning": 0.8661 + "Ties": 0.7819 } }, { @@ -4219,16 +4205,16 @@ "name": "GPT-4o mini 2024-07-18", "developer": "OpenAI", "scores": { - "Score": 0.8007, + "Score": 0.5796, + "Chat": 0.9497, + "Chat Hard": 0.6075, + "Safety": 0.7667, + "Reasoning": 0.8374, "Factuality": 0.4105, "Precise IF": 0.3438, "Math": 0.5191, - "Safety": 0.8081, "Focus": 0.7414, - "Ties": 0.6962, - "Chat": 0.9497, - "Chat Hard": 0.6075, - "Reasoning": 0.8374 + "Ties": 0.6962 } }, { @@ -4280,17 +4266,17 @@ "name": "openbmb/UltraRM-13b", "developer": "openbmb", "scores": { - "Score": 0.4683, - "Chat": 0.9637, - "Chat Hard": 0.5548, - "Safety": 0.5089, - "Reasoning": 0.6244, - "Prior Sets (0.5 weight)": 0.7294, + "Score": 0.6903, "Factuality": 0.5063, "Precise IF": 0.3312, "Math": 0.5519, + "Safety": 0.5986, "Focus": 0.6081, - "Ties": 0.3036 + "Ties": 0.3036, + "Chat": 0.9637, + "Chat Hard": 0.5548, + "Reasoning": 0.6244, + "Prior Sets (0.5 weight)": 0.7294 } }, { @@ -4370,17 +4356,17 @@ "name": "sfairXC/FsfairX-LLaMA3-RM-v0.1", "developer": "sfairXC", "scores": { - "Score": 0.8338, + "Score": 0.6292, + "Chat": 0.9944, + "Chat Hard": 0.6513, + "Safety": 0.7667, + "Reasoning": 0.8644, + "Prior Sets (0.5 weight)": 0.7492, "Factuality": 0.5916, "Precise IF": 0.4188, "Math": 0.6284, - "Safety": 0.8676, "Focus": 0.7051, - "Ties": 0.6647, - "Chat": 0.9944, - "Chat Hard": 0.6513, - "Reasoning": 0.8644, - "Prior Sets (0.5 weight)": 0.7492 + "Ties": 0.6647 } }, { @@ -4492,17 +4478,17 @@ "name": "weqweasdas/RM-Gemma-2B", "developer": "weqweasdas", "scores": { - "Score": 0.6549, + "Score": 0.3057, + "Chat": 0.9441, + "Chat Hard": 0.4079, + "Safety": 0.3311, + "Reasoning": 0.7637, + "Prior Sets (0.5 weight)": 0.6652, "Factuality": 0.3705, "Precise IF": 0.2812, "Math": 0.4317, - "Safety": 0.4986, "Focus": 0.2343, - "Ties": 0.1851, - "Chat": 0.9441, - "Chat Hard": 0.4079, - "Reasoning": 0.7637, - "Prior Sets (0.5 weight)": 0.6652 + "Ties": 0.1851 } }, { @@ -4541,17 +4527,17 @@ "name": "weqweasdas/RM-Mistral-7B", "developer": "weqweasdas", "scores": { - "Score": 0.7982, + "Score": 0.596, + "Chat": 0.9665, + "Chat Hard": 0.6053, + "Safety": 0.6911, + "Reasoning": 0.7736, + "Prior Sets (0.5 weight)": 0.753, "Factuality": 0.5937, "Precise IF": 0.3438, "Math": 0.5956, - "Safety": 0.8703, "Focus": 0.7293, - "Ties": 0.6226, - "Chat": 0.9665, - "Chat Hard": 0.6053, - "Reasoning": 0.7736, - "Prior Sets (0.5 weight)": 0.753 + "Ties": 0.6226 } }, { @@ -4559,17 +4545,17 @@ "name": "weqweasdas/hh_rlhf_rm_open_llama_3b", "developer": "weqweasdas", "scores": { - "Score": 0.5027, + "Score": 0.2498, + "Chat": 0.8184, + "Chat Hard": 0.3728, + "Safety": 0.24, + "Reasoning": 0.3281, + "Prior Sets (0.5 weight)": 0.6564, "Factuality": 0.3642, "Precise IF": 0.275, "Math": 0.3497, - "Safety": 0.4149, "Focus": 0.2384, - "Ties": 0.0315, - "Chat": 0.8184, - "Chat Hard": 0.3728, - "Reasoning": 0.3281, - "Prior Sets (0.5 weight)": 0.6564 + "Ties": 0.0315 } } ] diff --git a/data/benchmarks/swe-bench.json b/data/benchmarks/swe-bench.json index c5ac3821d8cf33006a01a757ed86c4602272b367..88b6df39554e558930cdff771fd1ab7d2dd73e3a 100644 --- a/data/benchmarks/swe-bench.json +++ b/data/benchmarks/swe-bench.json @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "swe-bench": 0.5455 + "swe-bench": 0.57 } } ] diff --git a/data/benchmarks/tau-bench-2_airline.json b/data/benchmarks/tau-bench-2_airline.json index f12d28637a77b58341be6f902aadfc1d22527d1a..3829696a07bc037642c69851a891e0aeb0e5febf 100644 --- a/data/benchmarks/tau-bench-2_airline.json +++ b/data/benchmarks/tau-bench-2_airline.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/airline": 0.66 + "tau-bench-2/airline": 0.72 } }, { diff --git a/data/benchmarks/tau-bench-2_retail.json b/data/benchmarks/tau-bench-2_retail.json index 8567872e9ca3424b440aa66923f7ea7a010a7290..220f36fbae7a50d2d3974e0cd62ceacad728d0fd 100644 --- a/data/benchmarks/tau-bench-2_retail.json +++ b/data/benchmarks/tau-bench-2_retail.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/retail": 0.78 + "tau-bench-2/retail": 0.85 } }, { @@ -13,7 +13,7 @@ "name": "gemini-3-pro-preview", "developer": "Google", "scores": { - "tau-bench-2/retail": 0.73 + "tau-bench-2/retail": 0.7805 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/retail": 0.73 + "tau-bench-2/retail": 0.68 } } ] diff --git a/data/benchmarks/tau-bench-2_telecom.json b/data/benchmarks/tau-bench-2_telecom.json index 717a8c139daad73d2bba1920f3f4fdded08dd42b..5e2e97c5a63c814404bfd0e936bb7f41ce63593e 100644 --- a/data/benchmarks/tau-bench-2_telecom.json +++ b/data/benchmarks/tau-bench-2_telecom.json @@ -5,7 +5,7 @@ "name": "claude-opus-4-5", "developer": "Anthropic", "scores": { - "tau-bench-2/telecom": 0.84 + "tau-bench-2/telecom": 0.76 } }, { @@ -21,7 +21,7 @@ "name": "gpt-5.2-2025-12-11", "developer": "OpenAI", "scores": { - "tau-bench-2/telecom": 0.71 + "tau-bench-2/telecom": 0.5354 } } ] diff --git a/data/benchmarks/terminal-bench-2.0.json b/data/benchmarks/terminal-bench-2.0.json index 10b4a1e53538ae0245444d9327835077907520b5..eedfc30dd08733e57be031879da03991ca05bc76 100644 --- a/data/benchmarks/terminal-bench-2.0.json +++ b/data/benchmarks/terminal-bench-2.0.json @@ -5,7 +5,7 @@ "name": "Qwen 3 Coder 480B", "developer": "Alibaba", "scores": { - "terminal-bench-2.0": 27.2 + "terminal-bench-2.0": 25.4 } }, { @@ -13,7 +13,7 @@ "name": "Claude Haiku 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 13.9 + "terminal-bench-2.0": 27.5 } }, { @@ -21,7 +21,7 @@ "name": "Claude Opus 4.1", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 35.1 + "terminal-bench-2.0": 34.8 } }, { @@ -37,7 +37,7 @@ "name": "Claude Opus 4.6", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 62.9 + "terminal-bench-2.0": 74.7 } }, { @@ -45,7 +45,7 @@ "name": "Claude Sonnet 4.5", "developer": "Anthropic", "scores": { - "terminal-bench-2.0": 42.6 + "terminal-bench-2.0": 42.5 } }, { @@ -77,7 +77,7 @@ "name": "Gemini 3 Flash", "developer": "Google", "scores": { - "terminal-bench-2.0": 47.4 + "terminal-bench-2.0": 51.0 } }, { @@ -85,7 +85,7 @@ "name": "Gemini 3 Pro", "developer": "Google", "scores": { - "terminal-bench-2.0": 56.0 + "terminal-bench-2.0": 56.9 } }, { @@ -109,7 +109,7 @@ "name": "MiniMax M2.1", "developer": "MiniMax", "scores": { - "terminal-bench-2.0": 36.6 + "terminal-bench-2.0": 29.2 } }, { @@ -125,7 +125,7 @@ "name": "Kimi K2 Instruct", "developer": "Moonshot AI", "scores": { - "terminal-bench-2.0": 27.8 + "terminal-bench-2.0": 26.7 } }, { @@ -149,7 +149,7 @@ "name": "Multiple", "developer": "Multiple", "scores": { - "terminal-bench-2.0": 72.4 + "terminal-bench-2.0": 71.0 } }, { @@ -173,7 +173,7 @@ "name": "GPT-5-Mini", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 24.0 + "terminal-bench-2.0": 31.9 } }, { @@ -181,7 +181,7 @@ "name": "GPT-5-Nano", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 11.5 + "terminal-bench-2.0": 7.0 } }, { @@ -197,7 +197,7 @@ "name": "GPT-5.1-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 57.8 + "terminal-bench-2.0": 53.5 } }, { @@ -221,7 +221,7 @@ "name": "GPT-5.2", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 62.9 + "terminal-bench-2.0": 60.7 } }, { @@ -237,7 +237,7 @@ "name": "GPT-5.3-Codex", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 77.3 + "terminal-bench-2.0": 74.6 } }, { @@ -245,7 +245,7 @@ "name": "GPT-OSS-120B", "developer": "OpenAI", "scores": { - "terminal-bench-2.0": 18.7 + "terminal-bench-2.0": 14.2 } }, { @@ -285,7 +285,7 @@ "name": "GLM 4.7", "developer": "Z-AI", "scores": { - "terminal-bench-2.0": 33.3 + "terminal-bench-2.0": 33.4 } }, { diff --git a/data/developers.json b/data/developers.json index f579192a07283e4405b9950165cfa67af8d0dcda..013b8e244887bf6c63a6c680ce52cd50fedc7ed7 100644 --- a/data/developers.json +++ b/data/developers.json @@ -173,7 +173,7 @@ }, { "developer": "allenai", - "model_count": 162 + "model_count": 161 }, { "developer": "allknowingroger", @@ -1097,7 +1097,7 @@ }, { "developer": "icefog72", - "model_count": 62 + "model_count": 61 }, { "developer": "IDEA-CCNL", @@ -1917,7 +1917,7 @@ }, { "developer": "NousResearch", - "model_count": 18 + "model_count": 19 }, { "developer": "Novaciano", diff --git a/data/developers/abhishek.json b/data/developers/abhishek.json index 4a990668962b400e2663f06544d4986dc1420d7b..93669d3c565d1e56b0d776cadd09cc45571db057 100644 --- a/data/developers/abhishek.json +++ b/data/developers/abhishek.json @@ -7,12 +7,12 @@ "developer": "abhishek", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1952, - "hfopenllm_v2/BBH": 0.3127, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3584, - "hfopenllm_v2/MMLU-PRO": 0.1144 + "hfopenllm_v2/IFEval": 0.1957, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.365, + "hfopenllm_v2/MMLU-PRO": 0.1151 } }, { diff --git a/data/developers/adriszmar.json b/data/developers/adriszmar.json index 1f1d39916960942963a9c3c265196aea3657be38..acb90d745752909d8f96f323acb9caa9b19061ae 100644 --- a/data/developers/adriszmar.json +++ b/data/developers/adriszmar.json @@ -7,12 +7,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1746, - "hfopenllm_v2/BBH": 0.3126, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.245, - "hfopenllm_v2/MUSR": 0.4096, - "hfopenllm_v2/MMLU-PRO": 0.1087 + "hfopenllm_v2/IFEval": 0.1685, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.3963, + "hfopenllm_v2/MMLU-PRO": 0.1066 } } ] diff --git a/data/developers/ai2.json b/data/developers/ai2.json index 6ae2e91a5d501e1d313c59819f3bee806d5615b0..498b6154facf655073d48826dd02116d29e34e45 100644 --- a/data/developers/ai2.json +++ b/data/developers/ai2.json @@ -43,10 +43,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6924, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.3575, - "reward-bench/Safety": 0.7757 + "reward-bench/Score": 0.6895, + "reward-bench/Chat": 0.9385, + "reward-bench/Chat Hard": 0.3706, + "reward-bench/Safety": 0.7595 } }, { diff --git a/data/developers/akjindal53244.json b/data/developers/akjindal53244.json index 86acdd918ca996450aa81c0111cbad62da1b8b17..237ea0357d953fdc2d416f7c27241c406836e723 100644 --- a/data/developers/akjindal53244.json +++ b/data/developers/akjindal53244.json @@ -7,12 +7,12 @@ "developer": "akjindal53244", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8033, - "hfopenllm_v2/BBH": 0.5196, - "hfopenllm_v2/MATH Level 5": 0.1624, - "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/IFEval": 0.8051, + "hfopenllm_v2/BBH": 0.5189, + "hfopenllm_v2/MATH Level 5": 0.1722, + "hfopenllm_v2/GPQA": 0.3263, "hfopenllm_v2/MUSR": 0.4028, - "hfopenllm_v2/MMLU-PRO": 0.3812 + "hfopenllm_v2/MMLU-PRO": 0.3803 } } ] diff --git a/data/developers/alibaba.json b/data/developers/alibaba.json index 844401e341e9e7b7e568fb2e5cdb8d2481e4039c..8efe41024b181a324827f30ff358f0234de7b987 100644 --- a/data/developers/alibaba.json +++ b/data/developers/alibaba.json @@ -7,7 +7,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.2 + "terminal-bench-2.0/terminal-bench-2.0": 25.4 } }, { diff --git a/data/developers/allenai.json b/data/developers/allenai.json index 4d8014825fe4883af161e8da92543c33ea544e4b..1c85dbe2cfcc674683235ce774bcd2bc78253288 100644 --- a/data/developers/allenai.json +++ b/data/developers/allenai.json @@ -82,17 +82,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.649, - "reward-bench/Chat": 0.933, - "reward-bench/Chat Hard": 0.7785, - "reward-bench/Safety": 0.8267, - "reward-bench/Reasoning": 0.7886, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8463, "reward-bench/Factuality": 0.72, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.8851, "reward-bench/Focus": 0.8323, - "reward-bench/Ties": 0.5406 + "reward-bench/Ties": 0.5406, + "reward-bench/Chat": 0.933, + "reward-bench/Chat Hard": 0.7785, + "reward-bench/Reasoning": 0.7886, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -120,12 +120,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8291, - "hfopenllm_v2/BBH": 0.6164, - "hfopenllm_v2/MATH Level 5": 0.4502, + "hfopenllm_v2/IFEval": 0.8379, + "hfopenllm_v2/BBH": 0.6157, + "hfopenllm_v2/MATH Level 5": 0.3829, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4948, - "hfopenllm_v2/MMLU-PRO": 0.4645 + "hfopenllm_v2/MUSR": 0.4988, + "hfopenllm_v2/MMLU-PRO": 0.4656 } }, { @@ -162,17 +162,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8892, + "reward-bench/Score": 0.722, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Safety": 0.8689, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, - "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8308 } }, { @@ -181,12 +181,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8267, - "hfopenllm_v2/BBH": 0.405, - "hfopenllm_v2/MATH Level 5": 0.1964, - "hfopenllm_v2/GPQA": 0.2987, + "hfopenllm_v2/IFEval": 0.8255, + "hfopenllm_v2/BBH": 0.4061, + "hfopenllm_v2/MATH Level 5": 0.2115, + "hfopenllm_v2/GPQA": 0.297, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2827 + "hfopenllm_v2/MMLU-PRO": 0.2821 } }, { @@ -209,17 +209,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8431, + "reward-bench/Score": 0.687, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.86, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6397 } }, { @@ -1054,21 +1054,6 @@ "reward-bench/Ties": 0.3534 } }, - { - "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "developer": "allenai", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.5151, - "reward-bench/Factuality": 0.6484, - "reward-bench/Precise IF": 0.3312, - "reward-bench/Math": 0.5574, - "reward-bench/Safety": 0.7289, - "reward-bench/Focus": 0.4889, - "reward-bench/Ties": 0.3357 - } - }, { "id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", diff --git a/data/developers/anthropic.json b/data/developers/anthropic.json index e348f58bb2289a52a6840206a2b95ae3acade6c2..819a852f845410b5f7876451b1dcb93ba59f6701 100644 --- a/data/developers/anthropic.json +++ b/data/developers/anthropic.json @@ -371,17 +371,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.3711, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Safety": 0.595, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635, + "reward-bench/Score": 0.7289, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, + "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899 + "reward-bench/Ties": 0.0899, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635 } }, { @@ -436,16 +436,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.5744, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.8378, - "reward-bench/Reasoning": 0.7868, + "reward-bench/Score": 0.8008, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601 + "reward-bench/Ties": 0.5601, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Reasoning": 0.7868 } }, { @@ -525,7 +525,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 13.9 + "terminal-bench-2.0/terminal-bench-2.0": 27.5 } }, { @@ -650,12 +650,12 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.7, + "appworld_test_normal/appworld/test_normal": 0.64, "browsecompplus/browsecompplus": 0.61, "swe-bench/swe-bench": 0.6061, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, - "tau-bench-2_retail/tau-bench-2/retail": 0.78, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 + "tau-bench-2_airline/tau-bench-2/airline": 0.72, + "tau-bench-2_retail/tau-bench-2/retail": 0.85, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 } }, { @@ -664,7 +664,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.1 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -682,7 +682,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.9 + "terminal-bench-2.0/terminal-bench-2.0": 74.7 } }, { @@ -756,7 +756,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.6 + "terminal-bench-2.0/terminal-bench-2.0": 42.5 } }, { diff --git a/data/developers/boltmonkey.json b/data/developers/boltmonkey.json index d93c347b80733b690b666d77f058a8e36708eee2..93a1093c0cbb01b9a84ae22cf6063fc50d8793cb 100644 --- a/data/developers/boltmonkey.json +++ b/data/developers/boltmonkey.json @@ -21,12 +21,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.459, - "hfopenllm_v2/BBH": 0.5185, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.2743, - "hfopenllm_v2/MUSR": 0.4083, - "hfopenllm_v2/MMLU-PRO": 0.3631 + "hfopenllm_v2/IFEval": 0.7999, + "hfopenllm_v2/BBH": 0.5152, + "hfopenllm_v2/MATH Level 5": 0.1193, + "hfopenllm_v2/GPQA": 0.281, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3733 } }, { diff --git a/data/developers/bunnycore.json b/data/developers/bunnycore.json index 153064e2dea6c3626cb6deb17e2d3191b25ec406..069b8a9f0214975725e28f414b82b07e73072c0f 100644 --- a/data/developers/bunnycore.json +++ b/data/developers/bunnycore.json @@ -287,12 +287,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1775, - "hfopenllm_v2/BBH": 0.295, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1049 + "hfopenllm_v2/IFEval": 0.4652, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.1284, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3394, + "hfopenllm_v2/MMLU-PRO": 0.3152 } }, { diff --git a/data/developers/cir-ams.json b/data/developers/cir-ams.json index df9dcecb6f8fae5d901c2496b58813341797427e..09d0cc390a2584b96dbfe9a5acec25171fe175fa 100644 --- a/data/developers/cir-ams.json +++ b/data/developers/cir-ams.json @@ -7,17 +7,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } } ] diff --git a/data/developers/columbia-nlp.json b/data/developers/columbia-nlp.json index 11f5fed45eb39522787aef4e964f3d3e28d320c0..b04d1f97bca6939c03a53d86cd396214edf82f72 100644 --- a/data/developers/columbia-nlp.json +++ b/data/developers/columbia-nlp.json @@ -7,12 +7,12 @@ "developer": "Columbia-NLP", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3102, - "hfopenllm_v2/BBH": 0.3881, - "hfopenllm_v2/MATH Level 5": 0.0536, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.4081, - "hfopenllm_v2/MMLU-PRO": 0.1665 + "hfopenllm_v2/IFEval": 0.3278, + "hfopenllm_v2/BBH": 0.392, + "hfopenllm_v2/MATH Level 5": 0.0431, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.412, + "hfopenllm_v2/MMLU-PRO": 0.1666 } }, { diff --git a/data/developers/daemontatox.json b/data/developers/daemontatox.json index 3de1c87bc255ec29e11a7fcd9434ebc4d17ff27a..09a88a4f4f23ea30a1c40f4893201987d0954b30 100644 --- a/data/developers/daemontatox.json +++ b/data/developers/daemontatox.json @@ -231,12 +231,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4855, - "hfopenllm_v2/BBH": 0.6627, - "hfopenllm_v2/MATH Level 5": 0.4841, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4256, - "hfopenllm_v2/MMLU-PRO": 0.5542 + "hfopenllm_v2/IFEval": 0.3745, + "hfopenllm_v2/BBH": 0.6668, + "hfopenllm_v2/MATH Level 5": 0.4758, + "hfopenllm_v2/GPQA": 0.3943, + "hfopenllm_v2/MUSR": 0.4858, + "hfopenllm_v2/MMLU-PRO": 0.5593 } }, { diff --git a/data/developers/davielion.json b/data/developers/davielion.json index ffdc7de10295de8981ccb0c2da137caa37979e5b..d2a3ca1471ea06bca721c18d505591908b22252e 100644 --- a/data/developers/davielion.json +++ b/data/developers/davielion.json @@ -7,12 +7,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { @@ -49,12 +49,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1336, - "hfopenllm_v2/BBH": 0.2975, - "hfopenllm_v2/MATH Level 5": 0.0068, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.35, - "hfopenllm_v2/MMLU-PRO": 0.1128 + "hfopenllm_v2/IFEval": 0.1324, + "hfopenllm_v2/BBH": 0.2972, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3527, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { diff --git a/data/developers/deepmount00.json b/data/developers/deepmount00.json index e898074e4a61782e05002f7b47eb2ee0411313aa..5505c28134c7ac10ef3a27978dd2745253c793a1 100644 --- a/data/developers/deepmount00.json +++ b/data/developers/deepmount00.json @@ -63,12 +63,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5365, - "hfopenllm_v2/BBH": 0.517, - "hfopenllm_v2/MATH Level 5": 0.1707, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4487, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.7917, + "hfopenllm_v2/BBH": 0.5109, + "hfopenllm_v2/MATH Level 5": 0.1088, + "hfopenllm_v2/GPQA": 0.2878, + "hfopenllm_v2/MUSR": 0.4136, + "hfopenllm_v2/MMLU-PRO": 0.3876 } }, { diff --git a/data/developers/dfurman.json b/data/developers/dfurman.json index 7e28f4da929deb69b4028e08d70bcf7cf516d943..2947dc3ef503295f24886c787e729305dcebb026 100644 --- a/data/developers/dfurman.json +++ b/data/developers/dfurman.json @@ -35,12 +35,12 @@ "developer": "dfurman", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3, - "hfopenllm_v2/BBH": 0.3853, - "hfopenllm_v2/MATH Level 5": 0.0415, - "hfopenllm_v2/GPQA": 0.2617, - "hfopenllm_v2/MUSR": 0.3579, - "hfopenllm_v2/MMLU-PRO": 0.2281 + "hfopenllm_v2/IFEval": 0.2835, + "hfopenllm_v2/BBH": 0.3842, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3566, + "hfopenllm_v2/MMLU-PRO": 0.2298 } }, { diff --git a/data/developers/epistemeai.json b/data/developers/epistemeai.json index ae59684b2ae8edda61f6033386c80ae35c1570fc..11b25cd31eead1ba78e7fde536703493137e208b 100644 --- a/data/developers/epistemeai.json +++ b/data/developers/epistemeai.json @@ -231,12 +231,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7305, - "hfopenllm_v2/BBH": 0.4649, - "hfopenllm_v2/MATH Level 5": 0.1397, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3209, - "hfopenllm_v2/MMLU-PRO": 0.348 + "hfopenllm_v2/IFEval": 0.7207, + "hfopenllm_v2/BBH": 0.461, + "hfopenllm_v2/MATH Level 5": 0.1314, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3432, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { diff --git a/data/developers/etherll.json b/data/developers/etherll.json index 6a72dd37a279f4d76a8244a057714c910854bcd2..2be2455f8558b72fcbd342c72cc671c443f3155e 100644 --- a/data/developers/etherll.json +++ b/data/developers/etherll.json @@ -35,12 +35,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6106, - "hfopenllm_v2/BBH": 0.5347, - "hfopenllm_v2/MATH Level 5": 0.1548, - "hfopenllm_v2/GPQA": 0.3146, - "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.3752 + "hfopenllm_v2/IFEval": 0.4672, + "hfopenllm_v2/BBH": 0.5013, + "hfopenllm_v2/MATH Level 5": 0.0279, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.386, + "hfopenllm_v2/MMLU-PRO": 0.3482 } }, { diff --git a/data/developers/goekdeniz-guelmez.json b/data/developers/goekdeniz-guelmez.json index e903240eb11157a57a20b31274250b901066286d..e3a743ff451afc2206e344fc708fd30240018483 100644 --- a/data/developers/goekdeniz-guelmez.json +++ b/data/developers/goekdeniz-guelmez.json @@ -63,12 +63,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3472, - "hfopenllm_v2/BBH": 0.3268, - "hfopenllm_v2/MATH Level 5": 0.0891, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3262, - "hfopenllm_v2/MMLU-PRO": 0.1641 + "hfopenllm_v2/IFEval": 0.3417, + "hfopenllm_v2/BBH": 0.3292, + "hfopenllm_v2/MATH Level 5": 0.0023, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3249, + "hfopenllm_v2/MMLU-PRO": 0.1638 } }, { diff --git a/data/developers/google.json b/data/developers/google.json index a447ddbc58f3e10c0347885b529d40f4033f2cb3..d2c89d05cf05ad4bda3e0747fb0e85f41c1968c7 100644 --- a/data/developers/google.json +++ b/data/developers/google.json @@ -157,8 +157,6 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.47, - "ace/Gaming Score": 0.509, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.373, "apex-agents/Overall Mean Score": 0.341, @@ -166,6 +164,8 @@ "apex-agents/Management Consulting Pass@1": 0.124, "apex-agents/Corporate Law Pass@1": 0.239, "apex-agents/Corporate Lawyer Mean Score": 0.487, + "ace/Overall Score": 0.47, + "ace/Gaming Score": 0.509, "apex-v1/Overall Score": 0.643, "apex-v1/Consulting Score": 0.64, "apex-v1/Investment Banking Score": 0.63 @@ -861,7 +861,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 47.4 + "terminal-bench-2.0/terminal-bench-2.0": 51.0 } }, { @@ -870,7 +870,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 56.0 + "terminal-bench-2.0/terminal-bench-2.0": 56.9 } }, { @@ -879,8 +879,8 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.55, - "browsecompplus/browsecompplus": 0.3333, + "appworld_test_normal/appworld/test_normal": 0.36, + "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, "global-mmlu-lite/Culturally Agnostic": 0.9509, @@ -902,7 +902,7 @@ "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.71, "tau-bench-2_airline/tau-bench-2/airline": 0.68, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_retail/tau-bench-2/retail": 0.7805, "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 } }, diff --git a/data/developers/guilhermenaturaumana.json b/data/developers/guilhermenaturaumana.json index bc6bb765fac527e1a43413f17dc37f0bfd85ba71..42443971cc3612439e60073a685d6fc6ddf14fe4 100644 --- a/data/developers/guilhermenaturaumana.json +++ b/data/developers/guilhermenaturaumana.json @@ -7,12 +7,12 @@ "developer": "GuilhermeNaturaUmana", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4985, - "hfopenllm_v2/BBH": 0.5645, - "hfopenllm_v2/MATH Level 5": 0.2576, - "hfopenllm_v2/GPQA": 0.3003, - "hfopenllm_v2/MUSR": 0.4373, - "hfopenllm_v2/MMLU-PRO": 0.4429 + "hfopenllm_v2/IFEval": 0.4791, + "hfopenllm_v2/BBH": 0.5649, + "hfopenllm_v2/MATH Level 5": 0.25, + "hfopenllm_v2/GPQA": 0.2995, + "hfopenllm_v2/MUSR": 0.4439, + "hfopenllm_v2/MMLU-PRO": 0.4408 } } ] diff --git a/data/developers/huggingfacetb.json b/data/developers/huggingfacetb.json index 1df25de945ccaf98a06fb9f3b47618bba05613d9..912c5d4a2004a890736a0d6051d6fabe190615d4 100644 --- a/data/developers/huggingfacetb.json +++ b/data/developers/huggingfacetb.json @@ -161,12 +161,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } } ] diff --git a/data/developers/icefog72.json b/data/developers/icefog72.json index 0ea5ae912613ba6349dc7eed95da5f8d3afd5dbe..7e3fc1f7a7c9a270611dd8a76e50d06bd5496253 100644 --- a/data/developers/icefog72.json +++ b/data/developers/icefog72.json @@ -813,20 +813,6 @@ "hfopenllm_v2/MMLU-PRO": 0.3103 } }, - { - "id": "icefog72/IceSakeV6RP-7b", - "name": "IceSakeV6RP-7b", - "developer": "icefog72", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5033, - "hfopenllm_v2/BBH": 0.4976, - "hfopenllm_v2/MATH Level 5": 0.0619, - "hfopenllm_v2/GPQA": 0.2911, - "hfopenllm_v2/MUSR": 0.42, - "hfopenllm_v2/MMLU-PRO": 0.3093 - } - }, { "id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", diff --git a/data/developers/infly.json b/data/developers/infly.json index d497bf1e2632542284e99f21127cba81c8ed1b97..fe3f0dc6f7a4b2c08dd2895544fd05de4f16df3c 100644 --- a/data/developers/infly.json +++ b/data/developers/infly.json @@ -7,16 +7,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9511, + "reward-bench/Score": 0.7648, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Safety": 0.9644, + "reward-bench/Reasoning": 0.9912, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Reasoning": 0.9912 + "reward-bench/Ties": 0.8622 } } ] diff --git a/data/developers/internlm.json b/data/developers/internlm.json index fbcc7249ad65cfb92e1e09809d6765ef54a982e2..3995d91bd1934deb8a0118739d074676cf30299b 100644 --- a/data/developers/internlm.json +++ b/data/developers/internlm.json @@ -21,16 +21,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -71,16 +71,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5335, - "reward-bench/Chat": 0.9916, - "reward-bench/Chat Hard": 0.6952, - "reward-bench/Safety": 0.5956, - "reward-bench/Reasoning": 0.9453, + "reward-bench/Score": 0.8759, "reward-bench/Factuality": 0.4211, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5628, + "reward-bench/Safety": 0.8716, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.5164 + "reward-bench/Ties": 0.5164, + "reward-bench/Chat": 0.9916, + "reward-bench/Chat Hard": 0.6952, + "reward-bench/Reasoning": 0.9453 } }, { diff --git a/data/developers/jaspionjader.json b/data/developers/jaspionjader.json index 9d9d1e268e56a9945ae657deca0493de6a22ce3d..053d128582b4aa02040ae51cd0577288669f17e5 100644 --- a/data/developers/jaspionjader.json +++ b/data/developers/jaspionjader.json @@ -1477,12 +1477,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4418, - "hfopenllm_v2/BBH": 0.5406, - "hfopenllm_v2/MATH Level 5": 0.1352, - "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5419, + "hfopenllm_v2/MATH Level 5": 0.1292, + "hfopenllm_v2/GPQA": 0.3087, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.386 + "hfopenllm_v2/MMLU-PRO": 0.3854 } }, { diff --git a/data/developers/leroydyer.json b/data/developers/leroydyer.json index 119d29bab457804b8d64637d339b1f72d7389de3..7fdecb6a1e65adfe5948ae6f60f4bc236a803ac7 100644 --- a/data/developers/leroydyer.json +++ b/data/developers/leroydyer.json @@ -679,12 +679,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3036, - "hfopenllm_v2/BBH": 0.4575, + "hfopenllm_v2/IFEval": 0.3066, + "hfopenllm_v2/BBH": 0.4577, "hfopenllm_v2/MATH Level 5": 0.0446, - "hfopenllm_v2/GPQA": 0.3012, - "hfopenllm_v2/MUSR": 0.4253, - "hfopenllm_v2/MMLU-PRO": 0.2329 + "hfopenllm_v2/GPQA": 0.2995, + "hfopenllm_v2/MUSR": 0.4254, + "hfopenllm_v2/MMLU-PRO": 0.2318 } }, { @@ -707,12 +707,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3798, - "hfopenllm_v2/BBH": 0.4483, - "hfopenllm_v2/MATH Level 5": 0.04, - "hfopenllm_v2/GPQA": 0.3129, - "hfopenllm_v2/MUSR": 0.4148, - "hfopenllm_v2/MMLU-PRO": 0.2389 + "hfopenllm_v2/IFEval": 0.3579, + "hfopenllm_v2/BBH": 0.4477, + "hfopenllm_v2/MATH Level 5": 0.0423, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4134, + "hfopenllm_v2/MMLU-PRO": 0.2376 } }, { diff --git a/data/developers/lxzgordon.json b/data/developers/lxzgordon.json index 7f802cf733857054e01537f3ecf745a3fdb38a05..e4ace3cc8f7193c8c711403c980535ee73fdd6d3 100644 --- a/data/developers/lxzgordon.json +++ b/data/developers/lxzgordon.json @@ -20,16 +20,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7394, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Safety": 0.9178, - "reward-bench/Reasoning": 0.9698, + "reward-bench/Score": 0.9294, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, + "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653 + "reward-bench/Ties": 0.7653, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Reasoning": 0.9698 } } ] diff --git a/data/developers/magpie-align.json b/data/developers/magpie-align.json index 0dc0a43e89bb456a61006caa30051add55effb08..155a416e715db19186e6af681f6736ea9ed101d1 100644 --- a/data/developers/magpie-align.json +++ b/data/developers/magpie-align.json @@ -35,12 +35,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4027, - "hfopenllm_v2/BBH": 0.4789, - "hfopenllm_v2/MATH Level 5": 0.0461, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.3087, - "hfopenllm_v2/MMLU-PRO": 0.3001 + "hfopenllm_v2/IFEval": 0.4118, + "hfopenllm_v2/BBH": 0.4811, + "hfopenllm_v2/MATH Level 5": 0.034, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3047, + "hfopenllm_v2/MMLU-PRO": 0.3006 } }, { diff --git a/data/developers/microsoft.json b/data/developers/microsoft.json index 5b22c3abc624b7d65a825afffb8ded406b5eb203..ac7ddab4c9378bc179c9df1d82dca24687067465 100644 --- a/data/developers/microsoft.json +++ b/data/developers/microsoft.json @@ -225,12 +225,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5477, - "hfopenllm_v2/BBH": 0.5491, - "hfopenllm_v2/MATH Level 5": 0.1639, - "hfopenllm_v2/GPQA": 0.3322, - "hfopenllm_v2/MUSR": 0.4284, - "hfopenllm_v2/MMLU-PRO": 0.4022 + "hfopenllm_v2/IFEval": 0.5613, + "hfopenllm_v2/BBH": 0.5676, + "hfopenllm_v2/MATH Level 5": 0.1163, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.395, + "hfopenllm_v2/MMLU-PRO": 0.3866 } }, { @@ -341,12 +341,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0585, - "hfopenllm_v2/BBH": 0.6691, - "hfopenllm_v2/MATH Level 5": 0.3165, - "hfopenllm_v2/GPQA": 0.406, + "hfopenllm_v2/IFEval": 0.0488, + "hfopenllm_v2/BBH": 0.6703, + "hfopenllm_v2/MATH Level 5": 0.2787, + "hfopenllm_v2/GPQA": 0.401, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5287 + "hfopenllm_v2/MMLU-PRO": 0.5295 } }, { diff --git a/data/developers/minimax.json b/data/developers/minimax.json index 3eb98fb6a6609e5f1cc4d74a47ecc2b74aaa9bb0..b575a16ccd7fb272ceb4b3067b0a9e48f65cff08 100644 --- a/data/developers/minimax.json +++ b/data/developers/minimax.json @@ -25,7 +25,7 @@ "developer": "MiniMax", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.6 + "terminal-bench-2.0/terminal-bench-2.0": 29.2 } }, { diff --git a/data/developers/mistralai.json b/data/developers/mistralai.json index 168bb98ff1b313fc7d40f024df899fec3a02671f..14d2f60be9e807210d185f950d2f81080246d7d7 100644 --- a/data/developers/mistralai.json +++ b/data/developers/mistralai.json @@ -513,12 +513,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.667, - "hfopenllm_v2/BBH": 0.5213, - "hfopenllm_v2/MATH Level 5": 0.1435, - "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.3632, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.6283, + "hfopenllm_v2/BBH": 0.583, + "hfopenllm_v2/MATH Level 5": 0.2039, + "hfopenllm_v2/GPQA": 0.3331, + "hfopenllm_v2/MUSR": 0.4063, + "hfopenllm_v2/MMLU-PRO": 0.4099 } }, { @@ -718,12 +718,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2415, - "hfopenllm_v2/BBH": 0.5087, - "hfopenllm_v2/MATH Level 5": 0.102, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4321, - "hfopenllm_v2/MMLU-PRO": 0.385 + "hfopenllm_v2/IFEval": 0.2326, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.3205, + "hfopenllm_v2/MUSR": 0.4413, + "hfopenllm_v2/MMLU-PRO": 0.3871 } }, { diff --git a/data/developers/mlabonne.json b/data/developers/mlabonne.json index 2620a8c4e8931697abdcd44e4a4aae7c1e430da5..be86bd7fe732025f133b06dc7c412aa5e56b7119 100644 --- a/data/developers/mlabonne.json +++ b/data/developers/mlabonne.json @@ -161,12 +161,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7561, - "hfopenllm_v2/BBH": 0.5111, - "hfopenllm_v2/MATH Level 5": 0.0906, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3841 + "hfopenllm_v2/IFEval": 0.4162, + "hfopenllm_v2/BBH": 0.5124, + "hfopenllm_v2/MATH Level 5": 0.0853, + "hfopenllm_v2/GPQA": 0.3029, + "hfopenllm_v2/MUSR": 0.415, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { diff --git a/data/developers/moonshot_ai.json b/data/developers/moonshot_ai.json index 746185ce773a539bd025922ab856fdcf2f8a1d9f..d83f11402fcca0c39c33d09eed495f2aefd69384 100644 --- a/data/developers/moonshot_ai.json +++ b/data/developers/moonshot_ai.json @@ -7,7 +7,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.8 + "terminal-bench-2.0/terminal-bench-2.0": 26.7 } }, { diff --git a/data/developers/multiple.json b/data/developers/multiple.json index 34cdb844d495e12fd3a3820204fbda313306e211..e235ffc0287578be5fd9fdf3ba4e4e1b232b5df8 100644 --- a/data/developers/multiple.json +++ b/data/developers/multiple.json @@ -7,7 +7,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 72.4 + "terminal-bench-2.0/terminal-bench-2.0": 71.0 } } ] diff --git a/data/developers/nicolinho.json b/data/developers/nicolinho.json index 79bf445ae201aa8b9add0559d92e4abd4fd3bebb..551d4a5de698babd0e830b509f51bb11f4dd2ac7 100644 --- a/data/developers/nicolinho.json +++ b/data/developers/nicolinho.json @@ -7,16 +7,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7667, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Safety": 0.9578, - "reward-bench/Reasoning": 0.9826, + "reward-bench/Score": 0.9444, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321 + "reward-bench/Ties": 0.8321, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Reasoning": 0.9826 } }, { @@ -51,16 +51,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7074, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Safety": 0.9467, - "reward-bench/Reasoning": 0.9677, + "reward-bench/Score": 0.9314, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234 + "reward-bench/Ties": 0.7234, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Reasoning": 0.9677 } } ] diff --git a/data/developers/nousresearch.json b/data/developers/nousresearch.json index 68e17c3374e0831b38026cc5c7fe37546bb1fc55..5eca3534d830aded2e15a419e7392ebd605b4769 100644 --- a/data/developers/nousresearch.json +++ b/data/developers/nousresearch.json @@ -200,6 +200,20 @@ "hfopenllm_v2/MMLU-PRO": 0.232 } }, + { + "id": "NousResearch/Yarn-Llama-2-7b-128k", + "name": "Yarn-Llama-2-7b-128k", + "developer": "NousResearch", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1485, + "hfopenllm_v2/BBH": 0.3248, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.2601, + "hfopenllm_v2/MUSR": 0.3967, + "hfopenllm_v2/MMLU-PRO": 0.1791 + } + }, { "id": "NousResearch/Yarn-Llama-2-7b-64k", "name": "Yarn-Llama-2-7b-64k", diff --git a/data/developers/ontocord.json b/data/developers/ontocord.json index add41dbe183800ebed90c916e0b704c974dd50e7..c16f4cbacdda3485b721f459b079923a6793a670 100644 --- a/data/developers/ontocord.json +++ b/data/developers/ontocord.json @@ -273,12 +273,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { diff --git a/data/developers/openai.json b/data/developers/openai.json index 463a516253b52dd581eb26a0054c21d8f415cdc5..1d144bd538c9cee7f94220b8970cc16ed56fa6ae 100644 --- a/data/developers/openai.json +++ b/data/developers/openai.json @@ -772,16 +772,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.8673, + "reward-bench/Score": 0.6493, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.8619, + "reward-bench/Reasoning": 0.8661, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, - "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.8661 + "reward-bench/Ties": 0.7819 } }, { @@ -859,16 +859,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 } }, { @@ -911,9 +911,9 @@ "helm_capabilities/IFEval": 0.875, "helm_capabilities/WildBench": 0.857, "helm_capabilities/Omni-MATH": 0.647, - "livecodebenchpro/Hard Problems": 0.04225352112676056, - "livecodebenchpro/Medium Problems": 0.4084507042253521, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0423, + "livecodebenchpro/Medium Problems": 0.4085, + "livecodebenchpro/Easy Problems": 0.9014 } }, { @@ -931,7 +931,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 24.0 + "terminal-bench-2.0/terminal-bench-2.0": 31.9 } }, { @@ -954,7 +954,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 11.5 + "terminal-bench-2.0/terminal-bench-2.0": 7.0 } }, { @@ -986,7 +986,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 57.8 + "terminal-bench-2.0/terminal-bench-2.0": 53.5 } }, { @@ -1013,7 +1013,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.9 + "terminal-bench-2.0/terminal-bench-2.0": 60.7 } }, { @@ -1023,14 +1023,14 @@ "evaluator_relationship": null, "benchmark_scores": { "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.43, + "browsecompplus/browsecompplus": 0.26, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.5455, + "swe-bench/swe-bench": 0.57, "tau-bench-2_airline/tau-bench-2/airline": 0.6, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.71 + "tau-bench-2_retail/tau-bench-2/retail": 0.68, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, { @@ -1048,7 +1048,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 77.3 + "terminal-bench-2.0/terminal-bench-2.0": 74.6 } }, { @@ -1112,7 +1112,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.11267605633802817, "livecodebenchpro/Easy Problems": 0.6619718309859155, - "terminal-bench-2.0/terminal-bench-2.0": 18.7 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } }, { @@ -1233,9 +1233,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.014084507042253521, - "livecodebenchpro/Medium Problems": 0.30985915492957744, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0143, + "livecodebenchpro/Medium Problems": 0.2923, + "livecodebenchpro/Easy Problems": 0.8571 } }, { diff --git a/data/developers/openassistant.json b/data/developers/openassistant.json index 7e4cf042377b82fe85282ade845062094d3d9218..0d3e3f32693fa46354d7063e0615e0d45e8641c5 100644 --- a/data/developers/openassistant.json +++ b/data/developers/openassistant.json @@ -7,17 +7,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2653, - "reward-bench/Chat": 0.9246, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Safety": 0.3289, - "reward-bench/Reasoning": 0.5855, - "reward-bench/Prior Sets (0.5 weight)": 0.6801, + "reward-bench/Score": 0.615, "reward-bench/Factuality": 0.3979, "reward-bench/Precise IF": 0.2875, "reward-bench/Math": 0.377, + "reward-bench/Safety": 0.5446, "reward-bench/Focus": 0.1535, - "reward-bench/Ties": 0.047 + "reward-bench/Ties": 0.047, + "reward-bench/Chat": 0.9246, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Reasoning": 0.5855, + "reward-bench/Prior Sets (0.5 weight)": 0.6801 } }, { diff --git a/data/developers/openbmb.json b/data/developers/openbmb.json index d8dae84054074ce01b5c47fc58b69a148fdc99c0..d6a398c44a61b95d325c54de1768af4a13c46f1f 100644 --- a/data/developers/openbmb.json +++ b/data/developers/openbmb.json @@ -68,17 +68,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4683, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Safety": 0.5089, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294, + "reward-bench/Score": 0.6903, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, + "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036 + "reward-bench/Ties": 0.3036, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294 } } ] diff --git a/data/developers/pku-alignment.json b/data/developers/pku-alignment.json index 76e1f41b6171c4fd2a3d35d25df175a75a77a416..0ae80803f93980ecd8558a983e84a1646a1bd7d6 100644 --- a/data/developers/pku-alignment.json +++ b/data/developers/pku-alignment.json @@ -7,17 +7,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5798, + "reward-bench/Score": 0.3332, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Safety": 0.7589, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, - "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57 + "reward-bench/Ties": -0.01 } }, { @@ -26,17 +26,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.1606, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Safety": 0.1422, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993, + "reward-bench/Score": 0.4727, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, + "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993 } }, { @@ -64,17 +64,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2544, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.364, - "reward-bench/Safety": 0.3156, - "reward-bench/Reasoning": 0.6887, - "reward-bench/Prior Sets (0.5 weight)": 0.6171, + "reward-bench/Score": 0.6366, "reward-bench/Factuality": 0.2168, "reward-bench/Precise IF": 0.2562, "reward-bench/Math": 0.3825, + "reward-bench/Safety": 0.6041, "reward-bench/Focus": 0.2606, - "reward-bench/Ties": 0.0944 + "reward-bench/Ties": 0.0944, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.364, + "reward-bench/Reasoning": 0.6887, + "reward-bench/Prior Sets (0.5 weight)": 0.6171 } } ] diff --git a/data/developers/primeintellect.json b/data/developers/primeintellect.json index 160722785b06f80d9b220dee435fad1245d45495..674a0e3b141480d7e0d33d0a2fe9b205b710216f 100644 --- a/data/developers/primeintellect.json +++ b/data/developers/primeintellect.json @@ -8,11 +8,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.274, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.25, - "hfopenllm_v2/MUSR": 0.3753, - "hfopenllm_v2/MMLU-PRO": 0.112 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3339, + "hfopenllm_v2/MMLU-PRO": 0.1123 } }, { diff --git a/data/developers/princeton-nlp.json b/data/developers/princeton-nlp.json index 10775d33e9c59e099c9bb1a2b84aa429c8a82f44..c2ef64ee6de3b0368749e3cc226eadfed8e75e4e 100644 --- a/data/developers/princeton-nlp.json +++ b/data/developers/princeton-nlp.json @@ -49,12 +49,12 @@ "developer": "princeton-nlp", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3978, - "hfopenllm_v2/BBH": 0.4983, - "hfopenllm_v2/MATH Level 5": 0.0582, - "hfopenllm_v2/GPQA": 0.281, - "hfopenllm_v2/MUSR": 0.425, - "hfopenllm_v2/MMLU-PRO": 0.3246 + "hfopenllm_v2/IFEval": 0.5508, + "hfopenllm_v2/BBH": 0.5028, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.4266, + "hfopenllm_v2/MMLU-PRO": 0.3231 } }, { diff --git a/data/developers/prithivmlmods.json b/data/developers/prithivmlmods.json index 88e743a4279f310dec935aed3968189a78be084e..f14f3864a910968b11c756eb8300bd4bd3eea36e 100644 --- a/data/developers/prithivmlmods.json +++ b/data/developers/prithivmlmods.json @@ -63,12 +63,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6052, - "hfopenllm_v2/BBH": 0.6317, - "hfopenllm_v2/MATH Level 5": 0.4789, - "hfopenllm_v2/GPQA": 0.3742, - "hfopenllm_v2/MUSR": 0.486, - "hfopenllm_v2/MMLU-PRO": 0.5302 + "hfopenllm_v2/IFEval": 0.6064, + "hfopenllm_v2/BBH": 0.6296, + "hfopenllm_v2/MATH Level 5": 0.3708, + "hfopenllm_v2/GPQA": 0.3733, + "hfopenllm_v2/MUSR": 0.4873, + "hfopenllm_v2/MMLU-PRO": 0.5307 } }, { diff --git a/data/developers/qingy2019.json b/data/developers/qingy2019.json index 9e607d380bbf6837078a684f50c178b1c83bab9f..f93967ea7ea0a389f1379b7b85428b832b816fb9 100644 --- a/data/developers/qingy2019.json +++ b/data/developers/qingy2019.json @@ -49,12 +49,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { diff --git a/data/developers/quazim0t0.json b/data/developers/quazim0t0.json index 496082d204a36ed57a388cd540f10080c702a420..135f1135e53ac7da23b4b9457344787d94d6e680 100644 --- a/data/developers/quazim0t0.json +++ b/data/developers/quazim0t0.json @@ -105,12 +105,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6654, - "hfopenllm_v2/BBH": 0.6901, - "hfopenllm_v2/MATH Level 5": 0.4698, - "hfopenllm_v2/GPQA": 0.3331, - "hfopenllm_v2/MUSR": 0.431, - "hfopenllm_v2/MMLU-PRO": 0.5426 + "hfopenllm_v2/IFEval": 0.6718, + "hfopenllm_v2/BBH": 0.6891, + "hfopenllm_v2/MATH Level 5": 0.4985, + "hfopenllm_v2/GPQA": 0.3339, + "hfopenllm_v2/MUSR": 0.4323, + "hfopenllm_v2/MMLU-PRO": 0.5408 } }, { @@ -637,12 +637,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2922, - "hfopenllm_v2/BBH": 0.6559, - "hfopenllm_v2/MATH Level 5": 0.2545, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3929, - "hfopenllm_v2/MMLU-PRO": 0.5207 + "hfopenllm_v2/IFEval": 0.7016, + "hfopenllm_v2/BBH": 0.6942, + "hfopenllm_v2/MATH Level 5": 0.4116, + "hfopenllm_v2/GPQA": 0.3624, + "hfopenllm_v2/MUSR": 0.4571, + "hfopenllm_v2/MMLU-PRO": 0.5411 } }, { diff --git a/data/developers/qwen.json b/data/developers/qwen.json index 8996ee46a22c0193f6bf8577a319d9b66b681c8b..727516da8d72881c91634dd8fcd12326550fd1bf 100644 --- a/data/developers/qwen.json +++ b/data/developers/qwen.json @@ -775,12 +775,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3071, - "hfopenllm_v2/BBH": 0.3341, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3329, - "hfopenllm_v2/MMLU-PRO": 0.1697 + "hfopenllm_v2/IFEval": 0.3153, + "hfopenllm_v2/BBH": 0.3322, + "hfopenllm_v2/MATH Level 5": 0.1035, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3342, + "hfopenllm_v2/MMLU-PRO": 0.172 } }, { diff --git a/data/developers/ray2333.json b/data/developers/ray2333.json index 0d33fd83045f7f1f2a0352b101b8f653545ad511..535b872cfc093237ac0d0dc85564191f8ac0a72c 100644 --- a/data/developers/ray2333.json +++ b/data/developers/ray2333.json @@ -61,16 +61,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5966, - "reward-bench/Chat": 0.9302, - "reward-bench/Chat Hard": 0.7719, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.912, + "reward-bench/Score": 0.8839, "reward-bench/Factuality": 0.5305, "reward-bench/Precise IF": 0.3125, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.9216, "reward-bench/Focus": 0.7455, - "reward-bench/Ties": 0.4788 + "reward-bench/Ties": 0.4788, + "reward-bench/Chat": 0.9302, + "reward-bench/Chat Hard": 0.7719, + "reward-bench/Reasoning": 0.912 } }, { @@ -98,16 +98,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6766, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8618, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.9362, + "reward-bench/Score": 0.9154, "reward-bench/Factuality": 0.6274, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.5847, + "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.8929, - "reward-bench/Ties": 0.6824 + "reward-bench/Ties": 0.6824, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8618, + "reward-bench/Reasoning": 0.9362 } }, { @@ -116,17 +116,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8542, + "reward-bench/Score": 0.6089, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Safety": 0.7867, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309 + "reward-bench/Ties": 0.5981 } }, { diff --git a/data/developers/recoilme.json b/data/developers/recoilme.json index 862469c1653f0d1414e3f6dabd97ef04b6d4f9e2..50baa7f16a67b3ee26a355a9f5b41de814bc7e52 100644 --- a/data/developers/recoilme.json +++ b/data/developers/recoilme.json @@ -35,12 +35,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7592, - "hfopenllm_v2/BBH": 0.6026, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4099, - "hfopenllm_v2/MMLU-PRO": 0.4163 + "hfopenllm_v2/IFEval": 0.2747, + "hfopenllm_v2/BBH": 0.6031, + "hfopenllm_v2/MATH Level 5": 0.0831, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4686, + "hfopenllm_v2/MMLU-PRO": 0.4122 } }, { @@ -49,12 +49,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5761, - "hfopenllm_v2/BBH": 0.602, - "hfopenllm_v2/MATH Level 5": 0.1888, - "hfopenllm_v2/GPQA": 0.3372, - "hfopenllm_v2/MUSR": 0.4632, - "hfopenllm_v2/MMLU-PRO": 0.4039 + "hfopenllm_v2/IFEval": 0.7439, + "hfopenllm_v2/BBH": 0.5993, + "hfopenllm_v2/MATH Level 5": 0.0876, + "hfopenllm_v2/GPQA": 0.3238, + "hfopenllm_v2/MUSR": 0.4204, + "hfopenllm_v2/MMLU-PRO": 0.4072 } }, { diff --git a/data/developers/replete-ai.json b/data/developers/replete-ai.json index 0f038b31b0c0cf28a02b6c25fb4fd9bd374c118c..dbb06f00736a7fcddf76b24a5c7673e098ecab57 100644 --- a/data/developers/replete-ai.json +++ b/data/developers/replete-ai.json @@ -91,12 +91,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { diff --git a/data/developers/sfairxc.json b/data/developers/sfairxc.json index 0f83a5aa819fcf69906563e63fd5c5a7cc1f8ce0..f511504fa472f754b653d8fff6432038f2fa5642 100644 --- a/data/developers/sfairxc.json +++ b/data/developers/sfairxc.json @@ -7,17 +7,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8338, + "reward-bench/Score": 0.6292, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492 + "reward-bench/Ties": 0.6647 } } ] diff --git a/data/developers/shikaichen.json b/data/developers/shikaichen.json index 6162cba6dec7eaed27af88272ffb98342af2522b..6502ef5b0e02cc85862537458f25816eb0826b7d 100644 --- a/data/developers/shikaichen.json +++ b/data/developers/shikaichen.json @@ -7,16 +7,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } } ] diff --git a/data/developers/skywork.json b/data/developers/skywork.json index 310fc474921f3ff87a2e73d5f74077c21e25d2fe..15f843864c8bce10620ff0dcdbde98d85af98ac4 100644 --- a/data/developers/skywork.json +++ b/data/developers/skywork.json @@ -47,16 +47,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7576, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Safety": 0.9422, - "reward-bench/Reasoning": 0.9606, + "reward-bench/Score": 0.938, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, + "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261 + "reward-bench/Ties": 0.8261, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Reasoning": 0.9606 } }, { @@ -230,16 +230,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6885, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.875, - "reward-bench/Safety": 0.8911, - "reward-bench/Reasoning": 0.9176, + "reward-bench/Score": 0.9007, "reward-bench/Factuality": 0.6063, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6339, + "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7586 + "reward-bench/Ties": 0.7586, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.875, + "reward-bench/Reasoning": 0.9176 } } ] diff --git a/data/developers/sometimesanotion.json b/data/developers/sometimesanotion.json index b756bb971c60b6bcad2e85989edb63b8a758fe59..c020e85fc5c690d215cbd09f94a114e50c6a6936 100644 --- a/data/developers/sometimesanotion.json +++ b/data/developers/sometimesanotion.json @@ -749,12 +749,12 @@ "developer": "sometimesanotion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5367, - "hfopenllm_v2/BBH": 0.6561, - "hfopenllm_v2/MATH Level 5": 0.358, - "hfopenllm_v2/GPQA": 0.3867, - "hfopenllm_v2/MUSR": 0.474, - "hfopenllm_v2/MMLU-PRO": 0.5395 + "hfopenllm_v2/IFEval": 0.5278, + "hfopenllm_v2/BBH": 0.6557, + "hfopenllm_v2/MATH Level 5": 0.3119, + "hfopenllm_v2/GPQA": 0.3842, + "hfopenllm_v2/MUSR": 0.4754, + "hfopenllm_v2/MMLU-PRO": 0.5396 } }, { diff --git a/data/developers/tanliboy.json b/data/developers/tanliboy.json index daa17945601d8528cbfc7c883e4dc4315ea363fc..7b17e651b33310ae087a6456408950e1456ff623 100644 --- a/data/developers/tanliboy.json +++ b/data/developers/tanliboy.json @@ -7,12 +7,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4501, - "hfopenllm_v2/BBH": 0.5472, - "hfopenllm_v2/MATH Level 5": 0.0944, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4017, - "hfopenllm_v2/MMLU-PRO": 0.3792 + "hfopenllm_v2/IFEval": 0.1829, + "hfopenllm_v2/BBH": 0.5488, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.4056, + "hfopenllm_v2/MMLU-PRO": 0.3805 } }, { diff --git a/data/developers/valiantlabs.json b/data/developers/valiantlabs.json index a0fc2b3d81b7ddfe31ae4c63f4f370e45c708501..b0b3adff2288383cd6923301f5751d53ee8ba504 100644 --- a/data/developers/valiantlabs.json +++ b/data/developers/valiantlabs.json @@ -49,12 +49,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7168, - "hfopenllm_v2/BBH": 0.4911, - "hfopenllm_v2/MATH Level 5": 0.1533, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.3512, - "hfopenllm_v2/MMLU-PRO": 0.3663 + "hfopenllm_v2/IFEval": 0.3496, + "hfopenllm_v2/BBH": 0.4947, + "hfopenllm_v2/MATH Level 5": 0.1269, + "hfopenllm_v2/GPQA": 0.3037, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.3644 } }, { diff --git a/data/developers/weqweasdas.json b/data/developers/weqweasdas.json index fd061c03aa20db7f3ab1d7ab2e9b6a6249c65553..36aa4e34297ffab30b15be8c96f1b385a32071bf 100644 --- a/data/developers/weqweasdas.json +++ b/data/developers/weqweasdas.json @@ -7,17 +7,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5027, + "reward-bench/Score": 0.2498, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.24, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, - "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564 + "reward-bench/Ties": 0.0315 } }, { @@ -26,17 +26,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6549, + "reward-bench/Score": 0.3057, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.4079, + "reward-bench/Safety": 0.3311, + "reward-bench/Reasoning": 0.7637, + "reward-bench/Prior Sets (0.5 weight)": 0.6652, "reward-bench/Factuality": 0.3705, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.4317, - "reward-bench/Safety": 0.4986, "reward-bench/Focus": 0.2343, - "reward-bench/Ties": 0.1851, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.4079, - "reward-bench/Reasoning": 0.7637, - "reward-bench/Prior Sets (0.5 weight)": 0.6652 + "reward-bench/Ties": 0.1851 } }, { @@ -78,17 +78,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7982, + "reward-bench/Score": 0.596, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Safety": 0.6911, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753 + "reward-bench/Ties": 0.6226 } } ] diff --git a/data/developers/yam-peleg.json b/data/developers/yam-peleg.json index 3161f95274f1998cddb90c53a59dce4097dbe0fd..f415128b8e93253c12ae15436b68cf36d5f248bf 100644 --- a/data/developers/yam-peleg.json +++ b/data/developers/yam-peleg.json @@ -35,12 +35,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } } ] diff --git a/data/developers/z-ai.json b/data/developers/z-ai.json index d4ada9599b117d1b4893fef821933bb44c4fb1d9..25c72391d11bee4c4ccdbee1d0c2cdded93b09ab 100644 --- a/data/developers/z-ai.json +++ b/data/developers/z-ai.json @@ -7,7 +7,7 @@ "developer": "Z-AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 33.3 + "terminal-bench-2.0/terminal-bench-2.0": 33.4 } }, { diff --git a/data/models.json b/data/models.json index 2b93623cbdf71160760101030d0eb4a96cc29224..a4ab0a1994ca0a7faae8835c098ea436fc0ab99c 100644 --- a/data/models.json +++ b/data/models.json @@ -907,12 +907,12 @@ "developer": "abhishek", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1952, - "hfopenllm_v2/BBH": 0.3127, - "hfopenllm_v2/MATH Level 5": 0.0128, - "hfopenllm_v2/GPQA": 0.2592, - "hfopenllm_v2/MUSR": 0.3584, - "hfopenllm_v2/MMLU-PRO": 0.1144 + "hfopenllm_v2/IFEval": 0.1957, + "hfopenllm_v2/BBH": 0.3135, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2517, + "hfopenllm_v2/MUSR": 0.365, + "hfopenllm_v2/MMLU-PRO": 0.1151 } }, { @@ -1005,12 +1005,12 @@ "developer": "adriszmar", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1746, - "hfopenllm_v2/BBH": 0.3126, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.245, - "hfopenllm_v2/MUSR": 0.4096, - "hfopenllm_v2/MMLU-PRO": 0.1087 + "hfopenllm_v2/IFEval": 0.1685, + "hfopenllm_v2/BBH": 0.3124, + "hfopenllm_v2/MATH Level 5": 0.0015, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.3963, + "hfopenllm_v2/MMLU-PRO": 0.1066 } }, { @@ -1391,10 +1391,10 @@ "developer": "AI2", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6924, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.3575, - "reward-bench/Safety": 0.7757 + "reward-bench/Score": 0.6895, + "reward-bench/Chat": 0.9385, + "reward-bench/Chat Hard": 0.3706, + "reward-bench/Safety": 0.7595 } }, { @@ -2036,12 +2036,12 @@ "developer": "akjindal53244", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8033, - "hfopenllm_v2/BBH": 0.5196, - "hfopenllm_v2/MATH Level 5": 0.1624, - "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/IFEval": 0.8051, + "hfopenllm_v2/BBH": 0.5189, + "hfopenllm_v2/MATH Level 5": 0.1722, + "hfopenllm_v2/GPQA": 0.3263, "hfopenllm_v2/MUSR": 0.4028, - "hfopenllm_v2/MMLU-PRO": 0.3812 + "hfopenllm_v2/MMLU-PRO": 0.3803 } }, { @@ -2243,7 +2243,7 @@ "developer": "Alibaba", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.2 + "terminal-bench-2.0/terminal-bench-2.0": 25.4 } }, { @@ -2409,17 +2409,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.649, - "reward-bench/Chat": 0.933, - "reward-bench/Chat Hard": 0.7785, - "reward-bench/Safety": 0.8267, - "reward-bench/Reasoning": 0.7886, - "reward-bench/Prior Sets (0.5 weight)": 0.0, + "reward-bench/Score": 0.8463, "reward-bench/Factuality": 0.72, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.8851, "reward-bench/Focus": 0.8323, - "reward-bench/Ties": 0.5406 + "reward-bench/Ties": 0.5406, + "reward-bench/Chat": 0.933, + "reward-bench/Chat Hard": 0.7785, + "reward-bench/Reasoning": 0.7886, + "reward-bench/Prior Sets (0.5 weight)": 0.0 } }, { @@ -2447,12 +2447,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8291, - "hfopenllm_v2/BBH": 0.6164, - "hfopenllm_v2/MATH Level 5": 0.4502, + "hfopenllm_v2/IFEval": 0.8379, + "hfopenllm_v2/BBH": 0.6157, + "hfopenllm_v2/MATH Level 5": 0.3829, "hfopenllm_v2/GPQA": 0.3733, - "hfopenllm_v2/MUSR": 0.4948, - "hfopenllm_v2/MMLU-PRO": 0.4645 + "hfopenllm_v2/MUSR": 0.4988, + "hfopenllm_v2/MMLU-PRO": 0.4656 } }, { @@ -2489,17 +2489,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8892, + "reward-bench/Score": 0.722, + "reward-bench/Chat": 0.9693, + "reward-bench/Chat Hard": 0.8268, + "reward-bench/Safety": 0.8689, + "reward-bench/Reasoning": 0.8583, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.8084, "reward-bench/Precise IF": 0.3688, "reward-bench/Math": 0.6776, - "reward-bench/Safety": 0.9027, "reward-bench/Focus": 0.7778, - "reward-bench/Ties": 0.8308, - "reward-bench/Chat": 0.9693, - "reward-bench/Chat Hard": 0.8268, - "reward-bench/Reasoning": 0.8583, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.8308 } }, { @@ -2508,12 +2508,12 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.8267, - "hfopenllm_v2/BBH": 0.405, - "hfopenllm_v2/MATH Level 5": 0.1964, - "hfopenllm_v2/GPQA": 0.2987, + "hfopenllm_v2/IFEval": 0.8255, + "hfopenllm_v2/BBH": 0.4061, + "hfopenllm_v2/MATH Level 5": 0.2115, + "hfopenllm_v2/GPQA": 0.297, "hfopenllm_v2/MUSR": 0.4175, - "hfopenllm_v2/MMLU-PRO": 0.2827 + "hfopenllm_v2/MMLU-PRO": 0.2821 } }, { @@ -2536,17 +2536,17 @@ "developer": "allenai", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8431, + "reward-bench/Score": 0.687, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.86, + "reward-bench/Reasoning": 0.7898, + "reward-bench/Prior Sets (0.5 weight)": 0.0, "reward-bench/Factuality": 0.7516, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.8545, - "reward-bench/Ties": 0.6397, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.7898, - "reward-bench/Prior Sets (0.5 weight)": 0.0 + "reward-bench/Ties": 0.6397 } }, { @@ -3381,21 +3381,6 @@ "reward-bench/Ties": 0.3534 } }, - { - "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "developer": "allenai", - "evaluator_relationship": null, - "benchmark_scores": { - "reward-bench/Score": 0.5151, - "reward-bench/Factuality": 0.6484, - "reward-bench/Precise IF": 0.3312, - "reward-bench/Math": 0.5574, - "reward-bench/Safety": 0.7289, - "reward-bench/Focus": 0.4889, - "reward-bench/Ties": 0.3357 - } - }, { "id": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", "name": "allenai/open_instruct_dev-rm_1e-6_2_5pctflipped__1__1743444636", @@ -7167,17 +7152,17 @@ "helm_mmlu/Virology": 0.542, "helm_mmlu/World Religions": 0.871, "helm_mmlu/Mean win rate": 0.28, - "reward-bench/Score": 0.3711, - "reward-bench/Chat": 0.9274, - "reward-bench/Chat Hard": 0.5197, - "reward-bench/Safety": 0.595, - "reward-bench/Reasoning": 0.706, - "reward-bench/Prior Sets (0.5 weight)": 0.6635, + "reward-bench/Score": 0.7289, "reward-bench/Factuality": 0.4042, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.3552, + "reward-bench/Safety": 0.7953, "reward-bench/Focus": 0.501, - "reward-bench/Ties": 0.0899 + "reward-bench/Ties": 0.0899, + "reward-bench/Chat": 0.9274, + "reward-bench/Chat Hard": 0.5197, + "reward-bench/Reasoning": 0.706, + "reward-bench/Prior Sets (0.5 weight)": 0.6635 } }, { @@ -7232,16 +7217,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.901, "helm_mmlu/Mean win rate": 0.014, - "reward-bench/Score": 0.5744, - "reward-bench/Chat": 0.9469, - "reward-bench/Chat Hard": 0.6031, - "reward-bench/Safety": 0.8378, - "reward-bench/Reasoning": 0.7868, + "reward-bench/Score": 0.8008, "reward-bench/Factuality": 0.5389, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5137, + "reward-bench/Safety": 0.8662, "reward-bench/Focus": 0.6646, - "reward-bench/Ties": 0.5601 + "reward-bench/Ties": 0.5601, + "reward-bench/Chat": 0.9469, + "reward-bench/Chat Hard": 0.6031, + "reward-bench/Reasoning": 0.7868 } }, { @@ -7321,7 +7306,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 13.9 + "terminal-bench-2.0/terminal-bench-2.0": 27.5 } }, { @@ -7446,12 +7431,12 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.7, + "appworld_test_normal/appworld/test_normal": 0.64, "browsecompplus/browsecompplus": 0.61, "swe-bench/swe-bench": 0.6061, - "tau-bench-2_airline/tau-bench-2/airline": 0.66, - "tau-bench-2_retail/tau-bench-2/retail": 0.78, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.84 + "tau-bench-2_airline/tau-bench-2/airline": 0.72, + "tau-bench-2_retail/tau-bench-2/retail": 0.85, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.76 } }, { @@ -7460,7 +7445,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 35.1 + "terminal-bench-2.0/terminal-bench-2.0": 34.8 } }, { @@ -7478,7 +7463,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.9 + "terminal-bench-2.0/terminal-bench-2.0": 74.7 } }, { @@ -7552,7 +7537,7 @@ "developer": "Anthropic", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 42.6 + "terminal-bench-2.0/terminal-bench-2.0": 42.5 } }, { @@ -9843,12 +9828,12 @@ "developer": "BoltMonkey", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.459, - "hfopenllm_v2/BBH": 0.5185, - "hfopenllm_v2/MATH Level 5": 0.0937, - "hfopenllm_v2/GPQA": 0.2743, - "hfopenllm_v2/MUSR": 0.4083, - "hfopenllm_v2/MMLU-PRO": 0.3631 + "hfopenllm_v2/IFEval": 0.7999, + "hfopenllm_v2/BBH": 0.5152, + "hfopenllm_v2/MATH Level 5": 0.1193, + "hfopenllm_v2/GPQA": 0.281, + "hfopenllm_v2/MUSR": 0.4019, + "hfopenllm_v2/MMLU-PRO": 0.3733 } }, { @@ -10599,12 +10584,12 @@ "developer": "bunnycore", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1775, - "hfopenllm_v2/BBH": 0.295, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3647, - "hfopenllm_v2/MMLU-PRO": 0.1049 + "hfopenllm_v2/IFEval": 0.4652, + "hfopenllm_v2/BBH": 0.4531, + "hfopenllm_v2/MATH Level 5": 0.1284, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3394, + "hfopenllm_v2/MMLU-PRO": 0.3152 } }, { @@ -11828,17 +11813,17 @@ "developer": "CIR-AMS", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8172, + "reward-bench/Score": 0.5736, + "reward-bench/Chat": 0.9749, + "reward-bench/Chat Hard": 0.5724, + "reward-bench/Safety": 0.7178, + "reward-bench/Reasoning": 0.8775, + "reward-bench/Prior Sets (0.5 weight)": 0.7029, "reward-bench/Factuality": 0.5347, "reward-bench/Precise IF": 0.3563, "reward-bench/Math": 0.6066, - "reward-bench/Safety": 0.9014, "reward-bench/Focus": 0.5737, - "reward-bench/Ties": 0.6527, - "reward-bench/Chat": 0.9749, - "reward-bench/Chat Hard": 0.5724, - "reward-bench/Reasoning": 0.8775, - "reward-bench/Prior Sets (0.5 weight)": 0.7029 + "reward-bench/Ties": 0.6527 } }, { @@ -12852,12 +12837,12 @@ "developer": "Columbia-NLP", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3102, - "hfopenllm_v2/BBH": 0.3881, - "hfopenllm_v2/MATH Level 5": 0.0536, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.4081, - "hfopenllm_v2/MMLU-PRO": 0.1665 + "hfopenllm_v2/IFEval": 0.3278, + "hfopenllm_v2/BBH": 0.392, + "hfopenllm_v2/MATH Level 5": 0.0431, + "hfopenllm_v2/GPQA": 0.2492, + "hfopenllm_v2/MUSR": 0.412, + "hfopenllm_v2/MMLU-PRO": 0.1666 } }, { @@ -14338,12 +14323,12 @@ "developer": "Daemontatox", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4855, - "hfopenllm_v2/BBH": 0.6627, - "hfopenllm_v2/MATH Level 5": 0.4841, - "hfopenllm_v2/GPQA": 0.3096, - "hfopenllm_v2/MUSR": 0.4256, - "hfopenllm_v2/MMLU-PRO": 0.5542 + "hfopenllm_v2/IFEval": 0.3745, + "hfopenllm_v2/BBH": 0.6668, + "hfopenllm_v2/MATH Level 5": 0.4758, + "hfopenllm_v2/GPQA": 0.3943, + "hfopenllm_v2/MUSR": 0.4858, + "hfopenllm_v2/MMLU-PRO": 0.5593 } }, { @@ -15393,12 +15378,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1507, - "hfopenllm_v2/BBH": 0.293, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/IFEval": 0.1549, + "hfopenllm_v2/BBH": 0.2937, + "hfopenllm_v2/MATH Level 5": 0.006, + "hfopenllm_v2/GPQA": 0.2576, "hfopenllm_v2/MUSR": 0.3565, - "hfopenllm_v2/MMLU-PRO": 0.1125 + "hfopenllm_v2/MMLU-PRO": 0.1128 } }, { @@ -15435,12 +15420,12 @@ "developer": "DavieLion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1336, - "hfopenllm_v2/BBH": 0.2975, - "hfopenllm_v2/MATH Level 5": 0.0068, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.35, - "hfopenllm_v2/MMLU-PRO": 0.1128 + "hfopenllm_v2/IFEval": 0.1324, + "hfopenllm_v2/BBH": 0.2972, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.2643, + "hfopenllm_v2/MUSR": 0.3527, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { @@ -15729,12 +15714,12 @@ "developer": "DeepMount00", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5365, - "hfopenllm_v2/BBH": 0.517, - "hfopenllm_v2/MATH Level 5": 0.1707, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4487, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.7917, + "hfopenllm_v2/BBH": 0.5109, + "hfopenllm_v2/MATH Level 5": 0.1088, + "hfopenllm_v2/GPQA": 0.2878, + "hfopenllm_v2/MUSR": 0.4136, + "hfopenllm_v2/MMLU-PRO": 0.3876 } }, { @@ -16376,12 +16361,12 @@ "developer": "dfurman", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3, - "hfopenllm_v2/BBH": 0.3853, - "hfopenllm_v2/MATH Level 5": 0.0415, - "hfopenllm_v2/GPQA": 0.2617, - "hfopenllm_v2/MUSR": 0.3579, - "hfopenllm_v2/MMLU-PRO": 0.2281 + "hfopenllm_v2/IFEval": 0.2835, + "hfopenllm_v2/BBH": 0.3842, + "hfopenllm_v2/MATH Level 5": 0.0521, + "hfopenllm_v2/GPQA": 0.2609, + "hfopenllm_v2/MUSR": 0.3566, + "hfopenllm_v2/MMLU-PRO": 0.2298 } }, { @@ -20340,12 +20325,12 @@ "developer": "EpistemeAI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7305, - "hfopenllm_v2/BBH": 0.4649, - "hfopenllm_v2/MATH Level 5": 0.1397, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3209, - "hfopenllm_v2/MMLU-PRO": 0.348 + "hfopenllm_v2/IFEval": 0.7207, + "hfopenllm_v2/BBH": 0.461, + "hfopenllm_v2/MATH Level 5": 0.1314, + "hfopenllm_v2/GPQA": 0.2701, + "hfopenllm_v2/MUSR": 0.3432, + "hfopenllm_v2/MMLU-PRO": 0.3354 } }, { @@ -21040,12 +21025,12 @@ "developer": "Etherll", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6106, - "hfopenllm_v2/BBH": 0.5347, - "hfopenllm_v2/MATH Level 5": 0.1548, - "hfopenllm_v2/GPQA": 0.3146, - "hfopenllm_v2/MUSR": 0.3991, - "hfopenllm_v2/MMLU-PRO": 0.3752 + "hfopenllm_v2/IFEval": 0.4672, + "hfopenllm_v2/BBH": 0.5013, + "hfopenllm_v2/MATH Level 5": 0.0279, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.386, + "hfopenllm_v2/MMLU-PRO": 0.3482 } }, { @@ -23303,12 +23288,12 @@ "developer": "Goekdeniz-Guelmez", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3472, - "hfopenllm_v2/BBH": 0.3268, - "hfopenllm_v2/MATH Level 5": 0.0891, - "hfopenllm_v2/GPQA": 0.2517, - "hfopenllm_v2/MUSR": 0.3262, - "hfopenllm_v2/MMLU-PRO": 0.1641 + "hfopenllm_v2/IFEval": 0.3417, + "hfopenllm_v2/BBH": 0.3292, + "hfopenllm_v2/MATH Level 5": 0.0023, + "hfopenllm_v2/GPQA": 0.2576, + "hfopenllm_v2/MUSR": 0.3249, + "hfopenllm_v2/MMLU-PRO": 0.1638 } }, { @@ -23537,8 +23522,6 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "ace/Overall Score": 0.47, - "ace/Gaming Score": 0.509, "apex-agents/Overall Pass@1": 0.184, "apex-agents/Overall Pass@8": 0.373, "apex-agents/Overall Mean Score": 0.341, @@ -23546,6 +23529,8 @@ "apex-agents/Management Consulting Pass@1": 0.124, "apex-agents/Corporate Law Pass@1": 0.239, "apex-agents/Corporate Lawyer Mean Score": 0.487, + "ace/Overall Score": 0.47, + "ace/Gaming Score": 0.509, "apex-v1/Overall Score": 0.643, "apex-v1/Consulting Score": 0.64, "apex-v1/Investment Banking Score": 0.63 @@ -24241,7 +24226,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 47.4 + "terminal-bench-2.0/terminal-bench-2.0": 51.0 } }, { @@ -24250,7 +24235,7 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 56.0 + "terminal-bench-2.0/terminal-bench-2.0": 56.9 } }, { @@ -24259,8 +24244,8 @@ "developer": "Google", "evaluator_relationship": null, "benchmark_scores": { - "appworld_test_normal/appworld/test_normal": 0.55, - "browsecompplus/browsecompplus": 0.3333, + "appworld_test_normal/appworld/test_normal": 0.36, + "browsecompplus/browsecompplus": 0.48, "global-mmlu-lite/Global MMLU Lite": 0.9453, "global-mmlu-lite/Culturally Sensitive": 0.9397, "global-mmlu-lite/Culturally Agnostic": 0.9509, @@ -24282,7 +24267,7 @@ "global-mmlu-lite/Burmese": 0.9425, "swe-bench/swe-bench": 0.71, "tau-bench-2_airline/tau-bench-2/airline": 0.68, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, + "tau-bench-2_retail/tau-bench-2/retail": 0.7805, "tau-bench-2_telecom/tau-bench-2/telecom": 0.73 } }, @@ -25530,12 +25515,12 @@ "developer": "GuilhermeNaturaUmana", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4985, - "hfopenllm_v2/BBH": 0.5645, - "hfopenllm_v2/MATH Level 5": 0.2576, - "hfopenllm_v2/GPQA": 0.3003, - "hfopenllm_v2/MUSR": 0.4373, - "hfopenllm_v2/MMLU-PRO": 0.4429 + "hfopenllm_v2/IFEval": 0.4791, + "hfopenllm_v2/BBH": 0.5649, + "hfopenllm_v2/MATH Level 5": 0.25, + "hfopenllm_v2/GPQA": 0.2995, + "hfopenllm_v2/MUSR": 0.4439, + "hfopenllm_v2/MMLU-PRO": 0.4408 } }, { @@ -26814,12 +26799,12 @@ "developer": "HuggingFaceTB", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.083, - "hfopenllm_v2/BBH": 0.3053, - "hfopenllm_v2/MATH Level 5": 0.0083, - "hfopenllm_v2/GPQA": 0.2651, - "hfopenllm_v2/MUSR": 0.3423, - "hfopenllm_v2/MMLU-PRO": 0.1126 + "hfopenllm_v2/IFEval": 0.3842, + "hfopenllm_v2/BBH": 0.3144, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.255, + "hfopenllm_v2/MUSR": 0.3461, + "hfopenllm_v2/MMLU-PRO": 0.1117 } }, { @@ -28221,20 +28206,6 @@ "hfopenllm_v2/MMLU-PRO": 0.3103 } }, - { - "id": "icefog72/IceSakeV6RP-7b", - "name": "IceSakeV6RP-7b", - "developer": "icefog72", - "evaluator_relationship": null, - "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5033, - "hfopenllm_v2/BBH": 0.4976, - "hfopenllm_v2/MATH Level 5": 0.0619, - "hfopenllm_v2/GPQA": 0.2911, - "hfopenllm_v2/MUSR": 0.42, - "hfopenllm_v2/MMLU-PRO": 0.3093 - } - }, { "id": "icefog72/IceSakeV8RP-7b", "name": "IceSakeV8RP-7b", @@ -28507,16 +28478,16 @@ "developer": "infly", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9511, + "reward-bench/Score": 0.7648, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9101, + "reward-bench/Safety": 0.9644, + "reward-bench/Reasoning": 0.9912, "reward-bench/Factuality": 0.7411, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6995, - "reward-bench/Safety": 0.9365, "reward-bench/Focus": 0.903, - "reward-bench/Ties": 0.8622, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9101, - "reward-bench/Reasoning": 0.9912 + "reward-bench/Ties": 0.8622 } }, { @@ -28651,16 +28622,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.3902, - "reward-bench/Chat": 0.9358, - "reward-bench/Chat Hard": 0.6623, - "reward-bench/Safety": 0.4711, - "reward-bench/Reasoning": 0.8724, + "reward-bench/Score": 0.8217, "reward-bench/Factuality": 0.2758, "reward-bench/Precise IF": 0.3625, "reward-bench/Math": 0.4426, + "reward-bench/Safety": 0.8162, "reward-bench/Focus": 0.596, - "reward-bench/Ties": 0.1934 + "reward-bench/Ties": 0.1934, + "reward-bench/Chat": 0.9358, + "reward-bench/Chat Hard": 0.6623, + "reward-bench/Reasoning": 0.8724 } }, { @@ -28701,16 +28672,16 @@ "developer": "internlm", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5335, - "reward-bench/Chat": 0.9916, - "reward-bench/Chat Hard": 0.6952, - "reward-bench/Safety": 0.5956, - "reward-bench/Reasoning": 0.9453, + "reward-bench/Score": 0.8759, "reward-bench/Factuality": 0.4211, "reward-bench/Precise IF": 0.4, "reward-bench/Math": 0.5628, + "reward-bench/Safety": 0.8716, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.5164 + "reward-bench/Ties": 0.5164, + "reward-bench/Chat": 0.9916, + "reward-bench/Chat Hard": 0.6952, + "reward-bench/Reasoning": 0.9453 } }, { @@ -30623,12 +30594,12 @@ "developer": "jaspionjader", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4418, - "hfopenllm_v2/BBH": 0.5406, - "hfopenllm_v2/MATH Level 5": 0.1352, - "hfopenllm_v2/GPQA": 0.3062, + "hfopenllm_v2/IFEval": 0.4345, + "hfopenllm_v2/BBH": 0.5419, + "hfopenllm_v2/MATH Level 5": 0.1292, + "hfopenllm_v2/GPQA": 0.3087, "hfopenllm_v2/MUSR": 0.4277, - "hfopenllm_v2/MMLU-PRO": 0.386 + "hfopenllm_v2/MMLU-PRO": 0.3854 } }, { @@ -37984,12 +37955,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3036, - "hfopenllm_v2/BBH": 0.4575, + "hfopenllm_v2/IFEval": 0.3066, + "hfopenllm_v2/BBH": 0.4577, "hfopenllm_v2/MATH Level 5": 0.0446, - "hfopenllm_v2/GPQA": 0.3012, - "hfopenllm_v2/MUSR": 0.4253, - "hfopenllm_v2/MMLU-PRO": 0.2329 + "hfopenllm_v2/GPQA": 0.2995, + "hfopenllm_v2/MUSR": 0.4254, + "hfopenllm_v2/MMLU-PRO": 0.2318 } }, { @@ -38012,12 +37983,12 @@ "developer": "LeroyDyer", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3798, - "hfopenllm_v2/BBH": 0.4483, - "hfopenllm_v2/MATH Level 5": 0.04, - "hfopenllm_v2/GPQA": 0.3129, - "hfopenllm_v2/MUSR": 0.4148, - "hfopenllm_v2/MMLU-PRO": 0.2389 + "hfopenllm_v2/IFEval": 0.3579, + "hfopenllm_v2/BBH": 0.4477, + "hfopenllm_v2/MATH Level 5": 0.0423, + "hfopenllm_v2/GPQA": 0.3096, + "hfopenllm_v2/MUSR": 0.4134, + "hfopenllm_v2/MMLU-PRO": 0.2376 } }, { @@ -39569,16 +39540,16 @@ "developer": "LxzGordon", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7394, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8816, - "reward-bench/Safety": 0.9178, - "reward-bench/Reasoning": 0.9698, + "reward-bench/Score": 0.9294, "reward-bench/Factuality": 0.6884, "reward-bench/Precise IF": 0.45, "reward-bench/Math": 0.6393, + "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.9758, - "reward-bench/Ties": 0.7653 + "reward-bench/Ties": 0.7653, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8816, + "reward-bench/Reasoning": 0.9698 } }, { @@ -39741,12 +39712,12 @@ "developer": "Magpie-Align", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4027, - "hfopenllm_v2/BBH": 0.4789, - "hfopenllm_v2/MATH Level 5": 0.0461, - "hfopenllm_v2/GPQA": 0.2768, - "hfopenllm_v2/MUSR": 0.3087, - "hfopenllm_v2/MMLU-PRO": 0.3001 + "hfopenllm_v2/IFEval": 0.4118, + "hfopenllm_v2/BBH": 0.4811, + "hfopenllm_v2/MATH Level 5": 0.034, + "hfopenllm_v2/GPQA": 0.2752, + "hfopenllm_v2/MUSR": 0.3047, + "hfopenllm_v2/MMLU-PRO": 0.3006 } }, { @@ -43235,12 +43206,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5477, - "hfopenllm_v2/BBH": 0.5491, - "hfopenllm_v2/MATH Level 5": 0.1639, - "hfopenllm_v2/GPQA": 0.3322, - "hfopenllm_v2/MUSR": 0.4284, - "hfopenllm_v2/MMLU-PRO": 0.4022 + "hfopenllm_v2/IFEval": 0.5613, + "hfopenllm_v2/BBH": 0.5676, + "hfopenllm_v2/MATH Level 5": 0.1163, + "hfopenllm_v2/GPQA": 0.3196, + "hfopenllm_v2/MUSR": 0.395, + "hfopenllm_v2/MMLU-PRO": 0.3866 } }, { @@ -43351,12 +43322,12 @@ "developer": "microsoft", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0585, - "hfopenllm_v2/BBH": 0.6691, - "hfopenllm_v2/MATH Level 5": 0.3165, - "hfopenllm_v2/GPQA": 0.406, + "hfopenllm_v2/IFEval": 0.0488, + "hfopenllm_v2/BBH": 0.6703, + "hfopenllm_v2/MATH Level 5": 0.2787, + "hfopenllm_v2/GPQA": 0.401, "hfopenllm_v2/MUSR": 0.5034, - "hfopenllm_v2/MMLU-PRO": 0.5287 + "hfopenllm_v2/MMLU-PRO": 0.5295 } }, { @@ -43667,7 +43638,7 @@ "developer": "MiniMax", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 36.6 + "terminal-bench-2.0/terminal-bench-2.0": 29.2 } }, { @@ -44247,12 +44218,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.667, - "hfopenllm_v2/BBH": 0.5213, - "hfopenllm_v2/MATH Level 5": 0.1435, - "hfopenllm_v2/GPQA": 0.3238, - "hfopenllm_v2/MUSR": 0.3632, - "hfopenllm_v2/MMLU-PRO": 0.396 + "hfopenllm_v2/IFEval": 0.6283, + "hfopenllm_v2/BBH": 0.583, + "hfopenllm_v2/MATH Level 5": 0.2039, + "hfopenllm_v2/GPQA": 0.3331, + "hfopenllm_v2/MUSR": 0.4063, + "hfopenllm_v2/MMLU-PRO": 0.4099 } }, { @@ -44452,12 +44423,12 @@ "developer": "mistralai", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2415, - "hfopenllm_v2/BBH": 0.5087, - "hfopenllm_v2/MATH Level 5": 0.102, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4321, - "hfopenllm_v2/MMLU-PRO": 0.385 + "hfopenllm_v2/IFEval": 0.2326, + "hfopenllm_v2/BBH": 0.5098, + "hfopenllm_v2/MATH Level 5": 0.0937, + "hfopenllm_v2/GPQA": 0.3205, + "hfopenllm_v2/MUSR": 0.4413, + "hfopenllm_v2/MMLU-PRO": 0.3871 } }, { @@ -44758,12 +44729,12 @@ "developer": "mlabonne", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7561, - "hfopenllm_v2/BBH": 0.5111, - "hfopenllm_v2/MATH Level 5": 0.0906, - "hfopenllm_v2/GPQA": 0.3062, - "hfopenllm_v2/MUSR": 0.4019, - "hfopenllm_v2/MMLU-PRO": 0.3841 + "hfopenllm_v2/IFEval": 0.4162, + "hfopenllm_v2/BBH": 0.5124, + "hfopenllm_v2/MATH Level 5": 0.0853, + "hfopenllm_v2/GPQA": 0.3029, + "hfopenllm_v2/MUSR": 0.415, + "hfopenllm_v2/MMLU-PRO": 0.3802 } }, { @@ -44996,7 +44967,7 @@ "developer": "Moonshot AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 27.8 + "terminal-bench-2.0/terminal-bench-2.0": 26.7 } }, { @@ -45317,7 +45288,7 @@ "developer": "Multiple", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 72.4 + "terminal-bench-2.0/terminal-bench-2.0": 71.0 } }, { @@ -48273,16 +48244,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7667, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.9013, - "reward-bench/Safety": 0.9578, - "reward-bench/Reasoning": 0.9826, + "reward-bench/Score": 0.9444, "reward-bench/Factuality": 0.7853, "reward-bench/Precise IF": 0.3719, "reward-bench/Math": 0.6995, + "reward-bench/Safety": 0.927, "reward-bench/Focus": 0.9535, - "reward-bench/Ties": 0.8321 + "reward-bench/Ties": 0.8321, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.9013, + "reward-bench/Reasoning": 0.9826 } }, { @@ -48317,16 +48288,16 @@ "developer": "nicolinho", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7074, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.8684, - "reward-bench/Safety": 0.9467, - "reward-bench/Reasoning": 0.9677, + "reward-bench/Score": 0.9314, "reward-bench/Factuality": 0.6653, "reward-bench/Precise IF": 0.4062, "reward-bench/Math": 0.612, + "reward-bench/Safety": 0.9257, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7234 + "reward-bench/Ties": 0.7234, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.8684, + "reward-bench/Reasoning": 0.9677 } }, { @@ -49326,6 +49297,20 @@ "hfopenllm_v2/MMLU-PRO": 0.232 } }, + { + "id": "NousResearch/Yarn-Llama-2-7b-128k", + "name": "Yarn-Llama-2-7b-128k", + "developer": "NousResearch", + "evaluator_relationship": null, + "benchmark_scores": { + "hfopenllm_v2/IFEval": 0.1485, + "hfopenllm_v2/BBH": 0.3248, + "hfopenllm_v2/MATH Level 5": 0.0151, + "hfopenllm_v2/GPQA": 0.2601, + "hfopenllm_v2/MUSR": 0.3967, + "hfopenllm_v2/MMLU-PRO": 0.1791 + } + }, { "id": "NousResearch/Yarn-Llama-2-7b-64k", "name": "Yarn-Llama-2-7b-64k", @@ -50379,12 +50364,12 @@ "developer": "ontocord", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1162, - "hfopenllm_v2/BBH": 0.3184, - "hfopenllm_v2/MATH Level 5": 0.0076, - "hfopenllm_v2/GPQA": 0.2634, - "hfopenllm_v2/MUSR": 0.3447, - "hfopenllm_v2/MMLU-PRO": 0.1124 + "hfopenllm_v2/IFEval": 0.1128, + "hfopenllm_v2/BBH": 0.3171, + "hfopenllm_v2/MATH Level 5": 0.0113, + "hfopenllm_v2/GPQA": 0.2685, + "hfopenllm_v2/MUSR": 0.346, + "hfopenllm_v2/MMLU-PRO": 0.1129 } }, { @@ -51606,16 +51591,16 @@ "helm_mmlu/Virology": 0.578, "helm_mmlu/World Religions": 0.883, "helm_mmlu/Mean win rate": 0.52, - "reward-bench/Score": 0.8673, + "reward-bench/Score": 0.6493, + "reward-bench/Chat": 0.9609, + "reward-bench/Chat Hard": 0.761, + "reward-bench/Safety": 0.8619, + "reward-bench/Reasoning": 0.8661, "reward-bench/Factuality": 0.5684, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.623, - "reward-bench/Safety": 0.8811, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.7819, - "reward-bench/Chat": 0.9609, - "reward-bench/Chat Hard": 0.761, - "reward-bench/Reasoning": 0.8661 + "reward-bench/Ties": 0.7819 } }, { @@ -51693,16 +51678,16 @@ "helm_mmlu/Virology": 0.536, "helm_mmlu/World Religions": 0.86, "helm_mmlu/Mean win rate": 0.774, - "reward-bench/Score": 0.8007, + "reward-bench/Score": 0.5796, + "reward-bench/Chat": 0.9497, + "reward-bench/Chat Hard": 0.6075, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8374, "reward-bench/Factuality": 0.4105, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5191, - "reward-bench/Safety": 0.8081, "reward-bench/Focus": 0.7414, - "reward-bench/Ties": 0.6962, - "reward-bench/Chat": 0.9497, - "reward-bench/Chat Hard": 0.6075, - "reward-bench/Reasoning": 0.8374 + "reward-bench/Ties": 0.6962 } }, { @@ -51745,9 +51730,9 @@ "helm_capabilities/IFEval": 0.875, "helm_capabilities/WildBench": 0.857, "helm_capabilities/Omni-MATH": 0.647, - "livecodebenchpro/Hard Problems": 0.04225352112676056, - "livecodebenchpro/Medium Problems": 0.4084507042253521, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0423, + "livecodebenchpro/Medium Problems": 0.4085, + "livecodebenchpro/Easy Problems": 0.9014 } }, { @@ -51765,7 +51750,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 24.0 + "terminal-bench-2.0/terminal-bench-2.0": 31.9 } }, { @@ -51788,7 +51773,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 11.5 + "terminal-bench-2.0/terminal-bench-2.0": 7.0 } }, { @@ -51820,7 +51805,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 57.8 + "terminal-bench-2.0/terminal-bench-2.0": 53.5 } }, { @@ -51847,7 +51832,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 62.9 + "terminal-bench-2.0/terminal-bench-2.0": 60.7 } }, { @@ -51857,14 +51842,14 @@ "evaluator_relationship": null, "benchmark_scores": { "appworld_test_normal/appworld/test_normal": 0.0, - "browsecompplus/browsecompplus": 0.43, + "browsecompplus/browsecompplus": 0.26, "livecodebenchpro/Hard Problems": 0.1594, "livecodebenchpro/Medium Problems": 0.5211, "livecodebenchpro/Easy Problems": 0.9014, - "swe-bench/swe-bench": 0.5455, + "swe-bench/swe-bench": 0.57, "tau-bench-2_airline/tau-bench-2/airline": 0.6, - "tau-bench-2_retail/tau-bench-2/retail": 0.73, - "tau-bench-2_telecom/tau-bench-2/telecom": 0.71 + "tau-bench-2_retail/tau-bench-2/retail": 0.68, + "tau-bench-2_telecom/tau-bench-2/telecom": 0.5354 } }, { @@ -51882,7 +51867,7 @@ "developer": "OpenAI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 77.3 + "terminal-bench-2.0/terminal-bench-2.0": 74.6 } }, { @@ -51946,7 +51931,7 @@ "livecodebenchpro/Hard Problems": 0.0, "livecodebenchpro/Medium Problems": 0.11267605633802817, "livecodebenchpro/Easy Problems": 0.6619718309859155, - "terminal-bench-2.0/terminal-bench-2.0": 18.7 + "terminal-bench-2.0/terminal-bench-2.0": 14.2 } }, { @@ -52067,9 +52052,9 @@ "helm_capabilities/IFEval": 0.929, "helm_capabilities/WildBench": 0.854, "helm_capabilities/Omni-MATH": 0.72, - "livecodebenchpro/Hard Problems": 0.014084507042253521, - "livecodebenchpro/Medium Problems": 0.30985915492957744, - "livecodebenchpro/Easy Problems": 0.8873239436619719 + "livecodebenchpro/Hard Problems": 0.0143, + "livecodebenchpro/Medium Problems": 0.2923, + "livecodebenchpro/Easy Problems": 0.8571 } }, { @@ -52213,17 +52198,17 @@ "developer": "OpenAssistant", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2653, - "reward-bench/Chat": 0.9246, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Safety": 0.3289, - "reward-bench/Reasoning": 0.5855, - "reward-bench/Prior Sets (0.5 weight)": 0.6801, + "reward-bench/Score": 0.615, "reward-bench/Factuality": 0.3979, "reward-bench/Precise IF": 0.2875, "reward-bench/Math": 0.377, + "reward-bench/Safety": 0.5446, "reward-bench/Focus": 0.1535, - "reward-bench/Ties": 0.047 + "reward-bench/Ties": 0.047, + "reward-bench/Chat": 0.9246, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Reasoning": 0.5855, + "reward-bench/Prior Sets (0.5 weight)": 0.6801 } }, { @@ -52345,17 +52330,17 @@ "developer": "openbmb", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.4683, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.5548, - "reward-bench/Safety": 0.5089, - "reward-bench/Reasoning": 0.6244, - "reward-bench/Prior Sets (0.5 weight)": 0.7294, + "reward-bench/Score": 0.6903, "reward-bench/Factuality": 0.5063, "reward-bench/Precise IF": 0.3312, "reward-bench/Math": 0.5519, + "reward-bench/Safety": 0.5986, "reward-bench/Focus": 0.6081, - "reward-bench/Ties": 0.3036 + "reward-bench/Ties": 0.3036, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.5548, + "reward-bench/Reasoning": 0.6244, + "reward-bench/Prior Sets (0.5 weight)": 0.7294 } }, { @@ -53816,17 +53801,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5798, + "reward-bench/Score": 0.3332, + "reward-bench/Chat": 0.6173, + "reward-bench/Chat Hard": 0.4232, + "reward-bench/Safety": 0.7589, + "reward-bench/Reasoning": 0.5482, + "reward-bench/Prior Sets (0.5 weight)": 0.57, "reward-bench/Factuality": 0.3263, "reward-bench/Precise IF": 0.2313, "reward-bench/Math": 0.3989, - "reward-bench/Safety": 0.7351, "reward-bench/Focus": 0.2939, - "reward-bench/Ties": -0.01, - "reward-bench/Chat": 0.6173, - "reward-bench/Chat Hard": 0.4232, - "reward-bench/Reasoning": 0.5482, - "reward-bench/Prior Sets (0.5 weight)": 0.57 + "reward-bench/Ties": -0.01 } }, { @@ -53835,17 +53820,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.1606, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.2873, - "reward-bench/Safety": 0.1422, - "reward-bench/Reasoning": 0.346, - "reward-bench/Prior Sets (0.5 weight)": 0.5993, + "reward-bench/Score": 0.4727, "reward-bench/Factuality": 0.2105, "reward-bench/Precise IF": 0.2938, "reward-bench/Math": 0.2623, + "reward-bench/Safety": 0.3757, "reward-bench/Focus": 0.0646, - "reward-bench/Ties": -0.01 + "reward-bench/Ties": -0.01, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.2873, + "reward-bench/Reasoning": 0.346, + "reward-bench/Prior Sets (0.5 weight)": 0.5993 } }, { @@ -53873,17 +53858,17 @@ "developer": "PKU-Alignment", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.2544, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.364, - "reward-bench/Safety": 0.3156, - "reward-bench/Reasoning": 0.6887, - "reward-bench/Prior Sets (0.5 weight)": 0.6171, + "reward-bench/Score": 0.6366, "reward-bench/Factuality": 0.2168, "reward-bench/Precise IF": 0.2562, "reward-bench/Math": 0.3825, + "reward-bench/Safety": 0.6041, "reward-bench/Focus": 0.2606, - "reward-bench/Ties": 0.0944 + "reward-bench/Ties": 0.0944, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.364, + "reward-bench/Reasoning": 0.6887, + "reward-bench/Prior Sets (0.5 weight)": 0.6171 } }, { @@ -54158,11 +54143,11 @@ "evaluator_relationship": null, "benchmark_scores": { "hfopenllm_v2/IFEval": 0.1757, - "hfopenllm_v2/BBH": 0.274, + "hfopenllm_v2/BBH": 0.276, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.25, - "hfopenllm_v2/MUSR": 0.3753, - "hfopenllm_v2/MMLU-PRO": 0.112 + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.3339, + "hfopenllm_v2/MMLU-PRO": 0.1123 } }, { @@ -54241,12 +54226,12 @@ "developer": "princeton-nlp", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3978, - "hfopenllm_v2/BBH": 0.4983, - "hfopenllm_v2/MATH Level 5": 0.0582, - "hfopenllm_v2/GPQA": 0.281, - "hfopenllm_v2/MUSR": 0.425, - "hfopenllm_v2/MMLU-PRO": 0.3246 + "hfopenllm_v2/IFEval": 0.5508, + "hfopenllm_v2/BBH": 0.5028, + "hfopenllm_v2/MATH Level 5": 0.0529, + "hfopenllm_v2/GPQA": 0.2861, + "hfopenllm_v2/MUSR": 0.4266, + "hfopenllm_v2/MMLU-PRO": 0.3231 } }, { @@ -54969,12 +54954,12 @@ "developer": "prithivMLmods", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6052, - "hfopenllm_v2/BBH": 0.6317, - "hfopenllm_v2/MATH Level 5": 0.4789, - "hfopenllm_v2/GPQA": 0.3742, - "hfopenllm_v2/MUSR": 0.486, - "hfopenllm_v2/MMLU-PRO": 0.5302 + "hfopenllm_v2/IFEval": 0.6064, + "hfopenllm_v2/BBH": 0.6296, + "hfopenllm_v2/MATH Level 5": 0.3708, + "hfopenllm_v2/GPQA": 0.3733, + "hfopenllm_v2/MUSR": 0.4873, + "hfopenllm_v2/MMLU-PRO": 0.5307 } }, { @@ -56591,12 +56576,12 @@ "developer": "qingy2019", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6005, - "hfopenllm_v2/BBH": 0.6356, - "hfopenllm_v2/MATH Level 5": 0.2764, - "hfopenllm_v2/GPQA": 0.3691, + "hfopenllm_v2/IFEval": 0.6066, + "hfopenllm_v2/BBH": 0.635, + "hfopenllm_v2/MATH Level 5": 0.3716, + "hfopenllm_v2/GPQA": 0.3725, "hfopenllm_v2/MUSR": 0.4757, - "hfopenllm_v2/MMLU-PRO": 0.5339 + "hfopenllm_v2/MMLU-PRO": 0.5331 } }, { @@ -56997,12 +56982,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.6654, - "hfopenllm_v2/BBH": 0.6901, - "hfopenllm_v2/MATH Level 5": 0.4698, - "hfopenllm_v2/GPQA": 0.3331, - "hfopenllm_v2/MUSR": 0.431, - "hfopenllm_v2/MMLU-PRO": 0.5426 + "hfopenllm_v2/IFEval": 0.6718, + "hfopenllm_v2/BBH": 0.6891, + "hfopenllm_v2/MATH Level 5": 0.4985, + "hfopenllm_v2/GPQA": 0.3339, + "hfopenllm_v2/MUSR": 0.4323, + "hfopenllm_v2/MMLU-PRO": 0.5408 } }, { @@ -57529,12 +57514,12 @@ "developer": "Quazim0t0", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.2922, - "hfopenllm_v2/BBH": 0.6559, - "hfopenllm_v2/MATH Level 5": 0.2545, - "hfopenllm_v2/GPQA": 0.2659, - "hfopenllm_v2/MUSR": 0.3929, - "hfopenllm_v2/MMLU-PRO": 0.5207 + "hfopenllm_v2/IFEval": 0.7016, + "hfopenllm_v2/BBH": 0.6942, + "hfopenllm_v2/MATH Level 5": 0.4116, + "hfopenllm_v2/GPQA": 0.3624, + "hfopenllm_v2/MUSR": 0.4571, + "hfopenllm_v2/MMLU-PRO": 0.5411 } }, { @@ -58633,12 +58618,12 @@ "developer": "Qwen", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.3071, - "hfopenllm_v2/BBH": 0.3341, - "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2576, - "hfopenllm_v2/MUSR": 0.3329, - "hfopenllm_v2/MMLU-PRO": 0.1697 + "hfopenllm_v2/IFEval": 0.3153, + "hfopenllm_v2/BBH": 0.3322, + "hfopenllm_v2/MATH Level 5": 0.1035, + "hfopenllm_v2/GPQA": 0.2592, + "hfopenllm_v2/MUSR": 0.3342, + "hfopenllm_v2/MMLU-PRO": 0.172 } }, { @@ -59368,16 +59353,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5966, - "reward-bench/Chat": 0.9302, - "reward-bench/Chat Hard": 0.7719, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.912, + "reward-bench/Score": 0.8839, "reward-bench/Factuality": 0.5305, "reward-bench/Precise IF": 0.3125, "reward-bench/Math": 0.5902, + "reward-bench/Safety": 0.9216, "reward-bench/Focus": 0.7455, - "reward-bench/Ties": 0.4788 + "reward-bench/Ties": 0.4788, + "reward-bench/Chat": 0.9302, + "reward-bench/Chat Hard": 0.7719, + "reward-bench/Reasoning": 0.912 } }, { @@ -59405,16 +59390,16 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6766, - "reward-bench/Chat": 0.9553, - "reward-bench/Chat Hard": 0.8618, - "reward-bench/Safety": 0.9222, - "reward-bench/Reasoning": 0.9362, + "reward-bench/Score": 0.9154, "reward-bench/Factuality": 0.6274, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.5847, + "reward-bench/Safety": 0.9081, "reward-bench/Focus": 0.8929, - "reward-bench/Ties": 0.6824 + "reward-bench/Ties": 0.6824, + "reward-bench/Chat": 0.9553, + "reward-bench/Chat Hard": 0.8618, + "reward-bench/Reasoning": 0.9362 } }, { @@ -59423,17 +59408,17 @@ "developer": "Ray2333", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8542, + "reward-bench/Score": 0.6089, + "reward-bench/Chat": 0.986, + "reward-bench/Chat Hard": 0.6776, + "reward-bench/Safety": 0.7867, + "reward-bench/Reasoning": 0.9229, + "reward-bench/Prior Sets (0.5 weight)": 0.7309, "reward-bench/Factuality": 0.6189, "reward-bench/Precise IF": 0.3875, "reward-bench/Math": 0.5792, - "reward-bench/Safety": 0.8919, "reward-bench/Focus": 0.6828, - "reward-bench/Ties": 0.5981, - "reward-bench/Chat": 0.986, - "reward-bench/Chat Hard": 0.6776, - "reward-bench/Reasoning": 0.9229, - "reward-bench/Prior Sets (0.5 weight)": 0.7309 + "reward-bench/Ties": 0.5981 } }, { @@ -59525,12 +59510,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7592, - "hfopenllm_v2/BBH": 0.6026, - "hfopenllm_v2/MATH Level 5": 0.0529, - "hfopenllm_v2/GPQA": 0.3289, - "hfopenllm_v2/MUSR": 0.4099, - "hfopenllm_v2/MMLU-PRO": 0.4163 + "hfopenllm_v2/IFEval": 0.2747, + "hfopenllm_v2/BBH": 0.6031, + "hfopenllm_v2/MATH Level 5": 0.0831, + "hfopenllm_v2/GPQA": 0.3305, + "hfopenllm_v2/MUSR": 0.4686, + "hfopenllm_v2/MMLU-PRO": 0.4122 } }, { @@ -59539,12 +59524,12 @@ "developer": "recoilme", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5761, - "hfopenllm_v2/BBH": 0.602, - "hfopenllm_v2/MATH Level 5": 0.1888, - "hfopenllm_v2/GPQA": 0.3372, - "hfopenllm_v2/MUSR": 0.4632, - "hfopenllm_v2/MMLU-PRO": 0.4039 + "hfopenllm_v2/IFEval": 0.7439, + "hfopenllm_v2/BBH": 0.5993, + "hfopenllm_v2/MATH Level 5": 0.0876, + "hfopenllm_v2/GPQA": 0.3238, + "hfopenllm_v2/MUSR": 0.4204, + "hfopenllm_v2/MMLU-PRO": 0.4072 } }, { @@ -59707,12 +59692,12 @@ "developer": "Replete-AI", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.0905, - "hfopenllm_v2/BBH": 0.2985, + "hfopenllm_v2/IFEval": 0.0932, + "hfopenllm_v2/BBH": 0.2977, "hfopenllm_v2/MATH Level 5": 0.0, - "hfopenllm_v2/GPQA": 0.2534, - "hfopenllm_v2/MUSR": 0.3848, - "hfopenllm_v2/MMLU-PRO": 0.1158 + "hfopenllm_v2/GPQA": 0.2475, + "hfopenllm_v2/MUSR": 0.3941, + "hfopenllm_v2/MMLU-PRO": 0.1157 } }, { @@ -62478,17 +62463,17 @@ "developer": "sfairXC", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.8338, + "reward-bench/Score": 0.6292, + "reward-bench/Chat": 0.9944, + "reward-bench/Chat Hard": 0.6513, + "reward-bench/Safety": 0.7667, + "reward-bench/Reasoning": 0.8644, + "reward-bench/Prior Sets (0.5 weight)": 0.7492, "reward-bench/Factuality": 0.5916, "reward-bench/Precise IF": 0.4188, "reward-bench/Math": 0.6284, - "reward-bench/Safety": 0.8676, "reward-bench/Focus": 0.7051, - "reward-bench/Ties": 0.6647, - "reward-bench/Chat": 0.9944, - "reward-bench/Chat Hard": 0.6513, - "reward-bench/Reasoning": 0.8644, - "reward-bench/Prior Sets (0.5 weight)": 0.7492 + "reward-bench/Ties": 0.6647 } }, { @@ -62567,16 +62552,16 @@ "developer": "ShikaiChen", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.9499, + "reward-bench/Score": 0.7249, + "reward-bench/Chat": 0.9637, + "reward-bench/Chat Hard": 0.9079, + "reward-bench/Safety": 0.9222, + "reward-bench/Reasoning": 0.9903, "reward-bench/Factuality": 0.7558, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6448, - "reward-bench/Safety": 0.9378, "reward-bench/Focus": 0.9131, - "reward-bench/Ties": 0.7633, - "reward-bench/Chat": 0.9637, - "reward-bench/Chat Hard": 0.9079, - "reward-bench/Reasoning": 0.9903 + "reward-bench/Ties": 0.7633 } }, { @@ -63269,16 +63254,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7576, - "reward-bench/Chat": 0.9581, - "reward-bench/Chat Hard": 0.9145, - "reward-bench/Safety": 0.9422, - "reward-bench/Reasoning": 0.9606, + "reward-bench/Score": 0.938, "reward-bench/Factuality": 0.7368, "reward-bench/Precise IF": 0.4031, "reward-bench/Math": 0.7049, + "reward-bench/Safety": 0.9189, "reward-bench/Focus": 0.9323, - "reward-bench/Ties": 0.8261 + "reward-bench/Ties": 0.8261, + "reward-bench/Chat": 0.9581, + "reward-bench/Chat Hard": 0.9145, + "reward-bench/Reasoning": 0.9606 } }, { @@ -63452,16 +63437,16 @@ "developer": "Skywork", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6885, - "reward-bench/Chat": 0.8994, - "reward-bench/Chat Hard": 0.875, - "reward-bench/Safety": 0.8911, - "reward-bench/Reasoning": 0.9176, + "reward-bench/Score": 0.9007, "reward-bench/Factuality": 0.6063, "reward-bench/Precise IF": 0.35, "reward-bench/Math": 0.6339, + "reward-bench/Safety": 0.9108, "reward-bench/Focus": 0.8909, - "reward-bench/Ties": 0.7586 + "reward-bench/Ties": 0.7586, + "reward-bench/Chat": 0.8994, + "reward-bench/Chat Hard": 0.875, + "reward-bench/Reasoning": 0.9176 } }, { @@ -64308,12 +64293,12 @@ "developer": "sometimesanotion", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.5367, - "hfopenllm_v2/BBH": 0.6561, - "hfopenllm_v2/MATH Level 5": 0.358, - "hfopenllm_v2/GPQA": 0.3867, - "hfopenllm_v2/MUSR": 0.474, - "hfopenllm_v2/MMLU-PRO": 0.5395 + "hfopenllm_v2/IFEval": 0.5278, + "hfopenllm_v2/BBH": 0.6557, + "hfopenllm_v2/MATH Level 5": 0.3119, + "hfopenllm_v2/GPQA": 0.3842, + "hfopenllm_v2/MUSR": 0.4754, + "hfopenllm_v2/MMLU-PRO": 0.5396 } }, { @@ -66770,12 +66755,12 @@ "developer": "tanliboy", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.4501, - "hfopenllm_v2/BBH": 0.5472, - "hfopenllm_v2/MATH Level 5": 0.0944, - "hfopenllm_v2/GPQA": 0.3138, - "hfopenllm_v2/MUSR": 0.4017, - "hfopenllm_v2/MMLU-PRO": 0.3792 + "hfopenllm_v2/IFEval": 0.1829, + "hfopenllm_v2/BBH": 0.5488, + "hfopenllm_v2/MATH Level 5": 0.0, + "hfopenllm_v2/GPQA": 0.3104, + "hfopenllm_v2/MUSR": 0.4056, + "hfopenllm_v2/MMLU-PRO": 0.3805 } }, { @@ -70986,12 +70971,12 @@ "developer": "ValiantLabs", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.7168, - "hfopenllm_v2/BBH": 0.4911, - "hfopenllm_v2/MATH Level 5": 0.1533, - "hfopenllm_v2/GPQA": 0.2861, - "hfopenllm_v2/MUSR": 0.3512, - "hfopenllm_v2/MMLU-PRO": 0.3663 + "hfopenllm_v2/IFEval": 0.3496, + "hfopenllm_v2/BBH": 0.4947, + "hfopenllm_v2/MATH Level 5": 0.1269, + "hfopenllm_v2/GPQA": 0.3037, + "hfopenllm_v2/MUSR": 0.3959, + "hfopenllm_v2/MMLU-PRO": 0.3644 } }, { @@ -71672,17 +71657,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.5027, + "reward-bench/Score": 0.2498, + "reward-bench/Chat": 0.8184, + "reward-bench/Chat Hard": 0.3728, + "reward-bench/Safety": 0.24, + "reward-bench/Reasoning": 0.3281, + "reward-bench/Prior Sets (0.5 weight)": 0.6564, "reward-bench/Factuality": 0.3642, "reward-bench/Precise IF": 0.275, "reward-bench/Math": 0.3497, - "reward-bench/Safety": 0.4149, "reward-bench/Focus": 0.2384, - "reward-bench/Ties": 0.0315, - "reward-bench/Chat": 0.8184, - "reward-bench/Chat Hard": 0.3728, - "reward-bench/Reasoning": 0.3281, - "reward-bench/Prior Sets (0.5 weight)": 0.6564 + "reward-bench/Ties": 0.0315 } }, { @@ -71691,17 +71676,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.6549, + "reward-bench/Score": 0.3057, + "reward-bench/Chat": 0.9441, + "reward-bench/Chat Hard": 0.4079, + "reward-bench/Safety": 0.3311, + "reward-bench/Reasoning": 0.7637, + "reward-bench/Prior Sets (0.5 weight)": 0.6652, "reward-bench/Factuality": 0.3705, "reward-bench/Precise IF": 0.2812, "reward-bench/Math": 0.4317, - "reward-bench/Safety": 0.4986, "reward-bench/Focus": 0.2343, - "reward-bench/Ties": 0.1851, - "reward-bench/Chat": 0.9441, - "reward-bench/Chat Hard": 0.4079, - "reward-bench/Reasoning": 0.7637, - "reward-bench/Prior Sets (0.5 weight)": 0.6652 + "reward-bench/Ties": 0.1851 } }, { @@ -71743,17 +71728,17 @@ "developer": "weqweasdas", "evaluator_relationship": null, "benchmark_scores": { - "reward-bench/Score": 0.7982, + "reward-bench/Score": 0.596, + "reward-bench/Chat": 0.9665, + "reward-bench/Chat Hard": 0.6053, + "reward-bench/Safety": 0.6911, + "reward-bench/Reasoning": 0.7736, + "reward-bench/Prior Sets (0.5 weight)": 0.753, "reward-bench/Factuality": 0.5937, "reward-bench/Precise IF": 0.3438, "reward-bench/Math": 0.5956, - "reward-bench/Safety": 0.8703, "reward-bench/Focus": 0.7293, - "reward-bench/Ties": 0.6226, - "reward-bench/Chat": 0.9665, - "reward-bench/Chat Hard": 0.6053, - "reward-bench/Reasoning": 0.7736, - "reward-bench/Prior Sets (0.5 weight)": 0.753 + "reward-bench/Ties": 0.6226 } }, { @@ -73033,12 +73018,12 @@ "developer": "yam-peleg", "evaluator_relationship": null, "benchmark_scores": { - "hfopenllm_v2/IFEval": 0.1856, - "hfopenllm_v2/BBH": 0.4149, - "hfopenllm_v2/MATH Level 5": 0.0234, - "hfopenllm_v2/GPQA": 0.276, - "hfopenllm_v2/MUSR": 0.3765, - "hfopenllm_v2/MMLU-PRO": 0.2573 + "hfopenllm_v2/IFEval": 0.177, + "hfopenllm_v2/BBH": 0.3411, + "hfopenllm_v2/MATH Level 5": 0.031, + "hfopenllm_v2/GPQA": 0.2534, + "hfopenllm_v2/MUSR": 0.374, + "hfopenllm_v2/MMLU-PRO": 0.2529 } }, { @@ -75535,7 +75520,7 @@ "developer": "Z-AI", "evaluator_relationship": null, "benchmark_scores": { - "terminal-bench-2.0/terminal-bench-2.0": 33.3 + "terminal-bench-2.0/terminal-bench-2.0": 33.4 } }, { diff --git a/data/models/abhishek_autotrain-0tmgq-5tpbg.json b/data/models/abhishek_autotrain-0tmgq-5tpbg.json index ac8f1c0a411cfe237fa1ffe1aaf7d7772ddbfda3..b3c557de2c59adc76d997235e3428686e7c95bbc 100644 --- a/data/models/abhishek_autotrain-0tmgq-5tpbg.json +++ b/data/models/abhishek_autotrain-0tmgq-5tpbg.json @@ -5,7 +5,7 @@ "developer": "abhishek", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "0.135" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1957 + "score": 0.1952 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3135 + "score": 0.3127 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0128 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2592 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.365 + "score": 0.3584 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1151 + "score": 0.1144 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1952 + "score": 0.1957 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3127 + "score": 0.3135 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0128 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2517 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3584 + "score": 0.365 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1144 + "score": 0.1151 } } ], diff --git a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json index 776a7c662b5afb276bf059bb90896cee5cd968ec..a8a69697e53329e56da6b7741229b84b4f2eeb85 100644 --- a/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json +++ b/data/models/adriszmar_qaimath-qwen2.5-7b-ties.json @@ -5,7 +5,7 @@ "developer": "adriszmar", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1685 + "score": 0.1746 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3124 + "score": 0.3126 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0015 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2492 + "score": 0.245 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3963 + "score": 0.4096 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1066 + "score": 0.1087 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1746 + "score": 0.1685 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3126 + "score": 0.3124 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0015 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.245 + "score": 0.2492 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4096 + "score": 0.3963 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1087 + "score": 0.1066 } } ], diff --git a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json index f109864a284814ea86e7b02eb2f140eae70857b8..036d2144a17c2773cc45b3fcdc429ec4c85d29b1 100644 --- a/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json +++ b/data/models/ai2_tulu-2-7b-rm-v0-nectar-binarized-3.8m-check....json @@ -38,7 +38,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6905 + "score": 0.7058 }, "source_data": { "dataset_name": "RewardBench", @@ -56,7 +56,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9525 }, "source_data": { "dataset_name": "RewardBench", @@ -74,7 +74,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3947 }, "source_data": { "dataset_name": "RewardBench", @@ -92,7 +92,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7676 + "score": 0.7703 }, "source_data": { "dataset_name": "RewardBench", @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7004 + "score": 0.6905 }, "source_data": { "dataset_name": "RewardBench", @@ -152,7 +152,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9413 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -170,7 +170,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -188,7 +188,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7716 + "score": 0.7676 }, "source_data": { "dataset_name": "RewardBench", @@ -230,7 +230,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6945 + "score": 0.7004 }, "source_data": { "dataset_name": "RewardBench", @@ -248,7 +248,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9413 }, "source_data": { "dataset_name": "RewardBench", @@ -266,7 +266,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -284,7 +284,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7743 + "score": 0.7716 }, "source_data": { "dataset_name": "RewardBench", @@ -326,7 +326,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7008 + "score": 0.6945 }, "source_data": { "dataset_name": "RewardBench", @@ -362,7 +362,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3882 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -380,7 +380,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7743 }, "source_data": { "dataset_name": "RewardBench", @@ -422,7 +422,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6895 + "score": 0.6808 }, "source_data": { "dataset_name": "RewardBench", @@ -440,7 +440,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9385 + "score": 0.9302 }, "source_data": { "dataset_name": "RewardBench", @@ -458,7 +458,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3706 + "score": 0.3596 }, "source_data": { "dataset_name": "RewardBench", @@ -476,7 +476,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7595 + "score": 0.7527 }, "source_data": { "dataset_name": "RewardBench", @@ -518,7 +518,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6808 + "score": 0.7008 }, "source_data": { "dataset_name": "RewardBench", @@ -536,7 +536,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -554,7 +554,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3596 + "score": 0.3882 }, "source_data": { "dataset_name": "RewardBench", @@ -572,7 +572,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7527 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -710,7 +710,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7058 + "score": 0.6924 }, "source_data": { "dataset_name": "RewardBench", @@ -728,7 +728,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9525 + "score": 0.9441 }, "source_data": { "dataset_name": "RewardBench", @@ -746,7 +746,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3947 + "score": 0.3575 }, "source_data": { "dataset_name": "RewardBench", @@ -764,7 +764,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7703 + "score": 0.7757 }, "source_data": { "dataset_name": "RewardBench", @@ -806,7 +806,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6924 + "score": 0.6895 }, "source_data": { "dataset_name": "RewardBench", @@ -824,7 +824,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.9385 }, "source_data": { "dataset_name": "RewardBench", @@ -842,7 +842,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3575 + "score": 0.3706 }, "source_data": { "dataset_name": "RewardBench", @@ -860,7 +860,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7757 + "score": 0.7595 }, "source_data": { "dataset_name": "RewardBench", diff --git a/data/models/akjindal53244_llama-3.1-storm-8b.json b/data/models/akjindal53244_llama-3.1-storm-8b.json index cd06cbbbbe2ccc306c529154ea621a16fb9ebcf6..10080988b150f205c06f956e608326efa2dd3fb0 100644 --- a/data/models/akjindal53244_llama-3.1-storm-8b.json +++ b/data/models/akjindal53244_llama-3.1-storm-8b.json @@ -5,7 +5,7 @@ "developer": "akjindal53244", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8051 + "score": 0.8033 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5189 + "score": 0.5196 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1722 + "score": 0.1624 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.3096 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3803 + "score": 0.3812 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8033 + "score": 0.8051 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5196 + "score": 0.5189 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1624 + "score": 0.1722 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3263 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3812 + "score": 0.3803 } } ], diff --git a/data/models/alibaba_qwen-3-coder-480b.json b/data/models/alibaba_qwen-3-coder-480b.json index 4f06f3ef146c1e8e47df3955820bbce15d9ad09d..5529ecfec8fec832ba0fc670045bbad34374bc66 100644 --- a/data/models/alibaba_qwen-3-coder-480b.json +++ b/data/models/alibaba_qwen-3-coder-480b.json @@ -4,13 +4,13 @@ "id": "alibaba/qwen-3-coder-480b", "developer": "Alibaba", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Dakou Agent", + "agent_organization": "iflow" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-12-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 25.4, + "score": 27.2, "uncertainty": { "standard_error": { "value": 2.6 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/dakou-agent__qwen-3-coder-480b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__qwen-3-coder-480b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-28", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,7 +191,7 @@ "max_score": 100.0 }, "score_details": { - "score": 27.2, + "score": 25.4, "uncertainty": { "standard_error": { "value": 2.6 @@ -201,7 +201,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Dakou Agent\" -m \"Qwen 3 Coder 480B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Qwen 3 Coder 480B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json b/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json index 73411ab4fcdf3bfd9a55ece3006bf091682be221..18e9257ee4a295645c5d70b31ea729844e529c44 100644 --- a/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json +++ b/data/models/alibaba_qwen3-235b-a22b-instruct-2507.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/alibaba_qwen3-235b-a22b-instruct-2507/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/allenai_llama-3.1-8b-base-rm-rb2.json b/data/models/allenai_llama-3.1-8b-base-rm-rb2.json index b77566d88bfa5b7f18f1a64562aee1f65b27a08c..8e81065be5c6f388ddbf031b25c6ea031d346e21 100644 --- a/data/models/allenai_llama-3.1-8b-base-rm-rb2.json +++ b/data/models/allenai_llama-3.1-8b-base-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8463 + "score": 0.649 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.933 + "score": 0.72 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7785 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.612 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8851 + "score": 0.8267 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7886 + "score": 0.8323 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.5406 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-8B-Base-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.649 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.72 + "score": 0.8463 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.933 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.612 + "score": 0.7785 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8851 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8323 + "score": 0.7886 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json index 56c7eedcfadd2447274047e1824ef99e14cf4e28..3c470e6abd70f5f2bc9a8be57cc36f939ca7753c 100644 --- a/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-70b-sft-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.722 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8084 + "score": 0.8892 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3688 + "score": 0.9693 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.8268 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8689 + "score": 0.9027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7778 + "score": 0.8583 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8308 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-70B-SFT-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8892 + "score": 0.722 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9693 + "score": 0.8084 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8268 + "score": 0.3688 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6776 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9027 + "score": 0.8689 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8583 + "score": 0.7778 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.8308 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-70b.json b/data/models/allenai_llama-3.1-tulu-3-70b.json index dedd220cd13dda3e92225bd87465795dab977b4c..1db7ede127ee803b7ffd5960ed83ddaa8fe5c58c 100644 --- a/data/models/allenai_llama-3.1-tulu-3-70b.json +++ b/data/models/allenai_llama-3.1-tulu-3-70b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "70.554" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8379 + "score": 0.8291 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6157 + "score": 0.6164 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3829 + "score": 0.4502 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4988 + "score": 0.4948 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4656 + "score": 0.4645 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8291 + "score": 0.8379 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6164 + "score": 0.6157 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4502 + "score": 0.3829 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4948 + "score": 0.4988 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4645 + "score": 0.4656 } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json index 52eed39924ec31b815aba5be343333e3d61411ae..408cdcea425aaae910c976c1f57587d29fcf9c63 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b-dpo-rm-rb2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.687 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7516 + "score": 0.8431 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.86 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8545 + "score": 0.7898 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6397 + "score": 0.0 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", + "evaluation_id": "reward-bench-2/allenai_Llama-3.1-Tulu-3-8B-DPO-RM-RB2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8431 + "score": 0.687 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.7516 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.86 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7898 + "score": 0.8545 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.6397 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/allenai_llama-3.1-tulu-3-8b.json b/data/models/allenai_llama-3.1-tulu-3-8b.json index 53350f20b1f74556a0af6a9aa87ed77d181deaec..7de1c9431728784c04f1c32781631febc2ed7e32 100644 --- a/data/models/allenai_llama-3.1-tulu-3-8b.json +++ b/data/models/allenai_llama-3.1-tulu-3-8b.json @@ -5,7 +5,7 @@ "developer": "allenai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8255 + "score": 0.8267 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4061 + "score": 0.405 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2115 + "score": 0.1964 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.297 + "score": 0.2987 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2821 + "score": 0.2827 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8267 + "score": 0.8255 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.405 + "score": 0.4061 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1964 + "score": 0.2115 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2987 + "score": 0.297 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2827 + "score": 0.2821 } } ], diff --git a/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json b/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json deleted file mode 100644 index cedfd398cf32e477ed09f75303d207e32a02614e..0000000000000000000000000000000000000000 --- a/data/models/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094.json +++ /dev/null @@ -1,162 +0,0 @@ -{ - "model_info": { - "name": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "id": "allenai/open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094", - "developer": "allenai", - "additional_details": { - "model_type": "Seq. Classifier" - } - }, - "evaluations": [ - { - "evaluation_id": "reward-bench-2/allenai_open_instruct_dev-rm_1e-6_2_30pctflipped__1__1743325094/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ - { - "evaluation_name": "Score", - "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5151 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6484 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Precise IF", - "metric_config": { - "evaluation_description": "Precise Instruction Following score", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3312 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Math", - "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5574 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Safety", - "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.7289 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Focus", - "metric_config": { - "evaluation_description": "Focus score - measures response focus", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4889 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Ties", - "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3357 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - } - ] -} \ No newline at end of file diff --git a/data/models/anthropic_claude-3-5-haiku-20241022.json b/data/models/anthropic_claude-3-5-haiku-20241022.json index 97ab32cadbb5e110dfa409cbea1ee9b8c38d6850..9edb0c7484cd3b1eab61fa90716227281f955253 100644 --- a/data/models/anthropic_claude-3-5-haiku-20241022.json +++ b/data/models/anthropic_claude-3-5-haiku-20241022.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-5-haiku-20241022/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-7-sonnet-20250219.json b/data/models/anthropic_claude-3-7-sonnet-20250219.json index 6e02a0c482b2aed6381eb4f121bd334b0959f1b6..26744b8fcce13c56de7ede1b941e4f633dfc781a 100644 --- a/data/models/anthropic_claude-3-7-sonnet-20250219.json +++ b/data/models/anthropic_claude-3-7-sonnet-20250219.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/anthropic_claude-3-7-sonnet-20250219/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/anthropic_claude-3-haiku-20240307.json b/data/models/anthropic_claude-3-haiku-20240307.json index 52bb9959dd5613be4f3fe143bd6f01f97e2632c2..4a2b563f634d87e24d77a5003bd4befff9181e63 100644 --- a/data/models/anthropic_claude-3-haiku-20240307.json +++ b/data/models/anthropic_claude-3-haiku-20240307.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,109 +1925,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7289 + "score": 0.3711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9274 + "score": 0.4042 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5197 + "score": 0.2812 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3552 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7953 + "score": 0.595 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.706 + "score": 0.501 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6635 + "score": 0.0899 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -2035,10 +2053,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/anthropic_claude-3-haiku-20240307/1766412838.146816", + "evaluation_id": "reward-bench/Anthropic_claude-3-haiku-20240307/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2057,127 +2075,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3711 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4042 + "score": 0.7289 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.9274 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3552 + "score": 0.5197 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.595 + "score": 0.7953 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.501 + "score": 0.706 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0899 + "score": 0.6635 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/anthropic_claude-3-opus-20240229.json b/data/models/anthropic_claude-3-opus-20240229.json index 8f8b70dd804fcbb225ceb8111209d93e20e8bd0b..150503f1a5f91571d10a33f8965d4bd84f3ccf46 100644 --- a/data/models/anthropic_claude-3-opus-20240229.json +++ b/data/models/anthropic_claude-3-opus-20240229.json @@ -1903,10 +1903,10 @@ } }, { - "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", + "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1925,128 +1925,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8008 + "score": 0.5744 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9469 + "score": 0.5389 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8662 + "score": 0.5137 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7868 + "score": 0.8378 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/anthropic_claude-3-opus-20240229/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5744 + "score": 0.6646 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2055,111 +2031,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5389 + "score": 0.5601 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Anthropic_claude-3-opus-20240229/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.8008 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5137 + "score": 0.9469 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8378 + "score": 0.6031 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6646 + "score": 0.8662 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5601 + "score": 0.7868 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/anthropic_claude-haiku-4.5.json b/data/models/anthropic_claude-haiku-4.5.json index a51d1ecfb03be58e1a56ba7e5711a9bf62822a49..a44b80f840579ff8c5f06292af8a76c394e19084 100644 --- a/data/models/anthropic_claude-haiku-4.5.json +++ b/data/models/anthropic_claude-haiku-4.5.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 28.3, + "score": 35.5, "uncertainty": { "standard_error": { "value": 2.9 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.5, + "score": 13.9, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 35.5, + "score": 28.3, "uncertainty": { "standard_error": { "value": 2.9 @@ -275,7 +275,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-haiku-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-haiku-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 13.9, + "score": 27.5, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Haiku 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Haiku 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4-5.json b/data/models/anthropic_claude-opus-4-5.json index 1de3c90261aa34fd7e316493f0ec72792c7a6a7a..e58ae243e6675ffbe5718b2cd53dbb80e8f1ba8b 100644 --- a/data/models/anthropic_claude-opus-4-5.json +++ b/data/models/anthropic_claude-opus-4-5.json @@ -78,7 +78,7 @@ } }, { - "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -91,42 +91,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.61, + "score": 0.66, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.59", - "total_run_cost": "759.44", - "average_steps": "27.18", - "percent_finished": "1.0" + "average_agent_cost": "13.08", + "total_run_cost": "1308.38", + "average_steps": "49.69", + "percent_finished": "0.74" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -159,34 +159,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.49, + "score": 0.61, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "7.09", - "total_run_cost": "709.54", - "average_steps": "21.66", - "percent_finished": "0.93" + "average_agent_cost": "11.32", + "total_run_cost": "1132.47", + "average_steps": "21.99", + "percent_finished": "0.83" } }, "generation_config": { @@ -214,7 +214,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -227,19 +227,19 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, @@ -251,18 +251,18 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "11.32", - "total_run_cost": "1132.47", - "average_steps": "21.99", - "percent_finished": "0.83" + "average_agent_cost": "6.3", + "total_run_cost": "630.56", + "average_steps": "24.16", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -314,23 +314,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.7, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "13.08", - "total_run_cost": "1308.38", - "average_steps": "49.69", - "percent_finished": "0.74" + "average_agent_cost": "5.59", + "total_run_cost": "558.51", + "average_steps": "41.07", + "percent_finished": "0.82" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -342,8 +342,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -486,7 +486,7 @@ } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -499,42 +499,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.49, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "5.59", - "total_run_cost": "558.51", - "average_steps": "41.07", - "percent_finished": "0.82" + "average_agent_cost": "7.09", + "total_run_cost": "709.54", + "average_steps": "21.66", + "percent_finished": "0.93" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -546,8 +546,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -622,7 +622,7 @@ } }, { - "evaluation_id": "browsecompplus/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -659,9 +659,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "6.3", - "total_run_cost": "630.56", - "average_steps": "24.16", + "average_agent_cost": "7.59", + "total_run_cost": "759.44", + "average_steps": "27.18", "percent_finished": "1.0" } }, @@ -669,8 +669,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -682,15 +682,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -722,14 +722,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7423, + "score": 0.65, "uncertainty": { - "num_samples": 97 + "num_samples": 100 }, "details": { - "average_agent_cost": "5.6", - "total_run_cost": "543.62", - "average_steps": "31.76", + "average_agent_cost": "4.85", + "total_run_cost": "485.22", + "average_steps": "39.13", "percent_finished": "1.0" } }, @@ -737,8 +737,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -750,15 +750,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -790,14 +790,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6061, + "score": 0.8072, "uncertainty": { - "num_samples": 99 + "num_samples": 83 }, "details": { - "average_agent_cost": "3.97", - "total_run_cost": "393.16", - "average_steps": "43.44", + "average_agent_cost": "2.96", + "total_run_cost": "245.78", + "average_steps": "34.1", "percent_finished": "1.0" } }, @@ -805,8 +805,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -818,15 +818,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -858,14 +858,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8072, + "score": 0.7423, "uncertainty": { - "num_samples": 83 + "num_samples": 97 }, "details": { - "average_agent_cost": "2.96", - "total_run_cost": "245.78", - "average_steps": "34.1", + "average_agent_cost": "5.6", + "total_run_cost": "543.62", + "average_steps": "31.76", "percent_finished": "1.0" } }, @@ -873,8 +873,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -886,15 +886,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -926,14 +926,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.65, + "score": 0.6061, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "4.85", - "total_run_cost": "485.22", - "average_steps": "39.13", + "average_agent_cost": "3.97", + "total_run_cost": "393.16", + "average_steps": "43.44", "percent_finished": "1.0" } }, @@ -941,8 +941,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -954,8 +954,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1030,7 +1030,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1062,14 +1062,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.72, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.78", - "total_run_cost": "39.67", - "average_steps": "11.88", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1077,8 +1077,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1090,8 +1090,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1166,7 +1166,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1198,14 +1198,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.74, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "average_agent_cost": "0.72", + "total_run_cost": "36.55", + "average_steps": "12.22", "percent_finished": "1.0" } }, @@ -1213,8 +1213,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1226,15 +1226,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1266,14 +1266,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.74, + "score": 0.66, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.72", - "total_run_cost": "36.55", - "average_steps": "12.22", + "average_agent_cost": "0.47", + "total_run_cost": "24.23", + "average_steps": "10.0", "percent_finished": "1.0" } }, @@ -1281,8 +1281,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1294,15 +1294,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1315,33 +1315,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.72, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "0.78", + "total_run_cost": "39.67", + "average_steps": "11.88", "percent_finished": "1.0" } }, @@ -1349,8 +1349,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1362,15 +1362,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1402,14 +1402,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.83, + "score": 0.78, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.6", - "total_run_cost": "161.14", - "average_steps": "12.54", + "average_agent_cost": "0.67", + "total_run_cost": "68.24", + "average_steps": "11.71", "percent_finished": "1.0" } }, @@ -1417,8 +1417,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1430,15 +1430,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1470,14 +1470,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.83, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.67", - "total_run_cost": "68.24", - "average_steps": "11.71", + "average_agent_cost": "1.6", + "total_run_cost": "161.14", + "average_steps": "12.54", "percent_finished": "1.0" } }, @@ -1485,8 +1485,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1498,15 +1498,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1519,33 +1519,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.66, + "score": 0.78, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { "average_agent_cost": "0.47", - "total_run_cost": "24.23", - "average_steps": "10.0", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1574,7 +1574,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1606,14 +1606,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.85, + "score": 0.78, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "56.18", - "average_steps": "12.54", + "average_agent_cost": "0.47", + "total_run_cost": "48.01", + "average_steps": "11.33", "percent_finished": "1.0" } }, @@ -1621,8 +1621,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1634,15 +1634,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1674,14 +1674,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.78, + "score": 0.85, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.47", - "total_run_cost": "48.01", - "average_steps": "11.33", + "average_agent_cost": "0.55", + "total_run_cost": "56.18", + "average_steps": "12.54", "percent_finished": "1.0" } }, @@ -1689,8 +1689,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1702,15 +1702,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1742,14 +1742,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.58, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "1.06", + "total_run_cost": "114.62", + "average_steps": "13.77", "percent_finished": "1.0" } }, @@ -1757,8 +1757,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1770,15 +1770,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1810,14 +1810,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.06", - "total_run_cost": "114.62", - "average_steps": "13.77", + "average_agent_cost": "2.45", + "total_run_cost": "255.97", + "average_steps": "18.71", "percent_finished": "1.0" } }, @@ -1825,8 +1825,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1838,15 +1838,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1883,9 +1883,9 @@ "num_samples": 100 }, "details": { - "average_agent_cost": "2.45", - "total_run_cost": "255.97", - "average_steps": "18.71", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -1893,8 +1893,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1906,15 +1906,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1946,14 +1946,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.76, + "score": 0.84, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.92", - "total_run_cost": "102.01", - "average_steps": "17.22", + "average_agent_cost": "1.25", + "total_run_cost": "136.84", + "average_steps": "17.15", "percent_finished": "1.0" } }, @@ -1961,8 +1961,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1974,15 +1974,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__anthropic_claude-opus-4-5/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__anthropic_claude-opus-4-5/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2014,14 +2014,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.84, + "score": 0.76, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.25", - "total_run_cost": "136.84", - "average_steps": "17.15", + "average_agent_cost": "0.92", + "total_run_cost": "102.01", + "average_steps": "17.22", "percent_finished": "1.0" } }, @@ -2029,8 +2029,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2042,8 +2042,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } diff --git a/data/models/anthropic_claude-opus-4.1.json b/data/models/anthropic_claude-opus-4.1.json index b51256b4d5a95f9160a82aa03d45fc55ec0aba09..59416d90b1dbdb120915f9f1b242f981bce8e890 100644 --- a/data/models/anthropic_claude-opus-4.1.json +++ b/data/models/anthropic_claude-opus-4.1.json @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 35.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-opus-4.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.1, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Opus 4.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.5.json b/data/models/anthropic_claude-opus-4.5.json index 14b1dc3ce90d80ab30d23bbb2e060321876d554a..007d3ac2c88dec5dff72fba0205d56eeaed42464 100644 --- a/data/models/anthropic_claude-opus-4.5.json +++ b/data/models/anthropic_claude-opus-4.5.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.5", "developer": "Anthropic", "additional_details": { - "agent_name": "OpenCode", - "agent_organization": "Anomaly Innovations" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-12", + "evaluation_timestamp": "2025-11-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,11 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7 + "score": 57.8, + "uncertainty": { + "standard_error": { + "value": 2.5 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -64,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -78,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -102,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-22", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -111,17 +117,11 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, - "uncertainty": { - "standard_error": { - "value": 2.5 - }, - "num_samples": 435 - } + "score": 58.4 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -138,7 +138,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -152,7 +152,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/opencode__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -176,7 +176,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-01-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -185,17 +185,11 @@ "max_score": 100.0 }, "score_details": { - "score": 54.3, - "uncertainty": { - "standard_error": { - "value": 2.6 - }, - "num_samples": 435 - } + "score": 51.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +206,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenCode\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +220,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/goose__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +244,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,17 +253,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.9, + "score": 54.3, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +280,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Goose\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +294,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +318,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,11 +327,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4 + "score": 59.1, + "uncertainty": { + "standard_error": { + "value": 2.4 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -354,7 +354,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -368,7 +368,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -392,7 +392,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -401,17 +401,17 @@ "max_score": 100.0 }, "score_details": { - "score": 63.1, + "score": 51.9, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -428,7 +428,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -442,7 +442,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__claude-opus-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -466,7 +466,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -475,17 +475,17 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 63.1, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -502,7 +502,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Claude Opus 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-opus-4.6.json b/data/models/anthropic_claude-opus-4.6.json index 048dae2de3da40230747c994f11a1351dfec5a14..6de13fc0f3140d257d348a852063485bd1007995 100644 --- a/data/models/anthropic_claude-opus-4.6.json +++ b/data/models/anthropic_claude-opus-4.6.json @@ -4,13 +4,13 @@ "id": "anthropic/claude-opus-4.6", "developer": "Anthropic", "additional_details": { - "agent_name": "Terminus-KIRA", - "agent_organization": "KRAFTON AI" + "agent_name": "Claude Code", + "agent_organization": "Anthropic" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.7, + "score": 58.0, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2026-02-13", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 69.9, + "score": 66.5, "uncertainty": { "standard_error": { "value": 2.5 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-13", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.5, + "score": 71.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-23", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,11 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 66.9 + "score": 62.9, + "uncertainty": { + "standard_error": { + "value": 2.7 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-07", + "evaluation_timestamp": "2026-02-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,17 +339,11 @@ "max_score": 100.0 }, "score_details": { - "score": 58.0, - "uncertainty": { - "standard_error": { - "value": 2.9 - }, - "num_samples": 435 - } + "score": 66.9 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -374,7 +374,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/tongagents__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -398,7 +398,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-22", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -407,17 +407,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.9, + "score": 69.9, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -434,7 +434,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"TongAgents\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__claude-opus-4.6/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-kira__claude-opus-4.6/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,17 +481,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, + "score": 74.7, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Claude Opus 4.6\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus-KIRA\" -m \"Claude Opus 4.6\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/anthropic_claude-sonnet-4.5.json b/data/models/anthropic_claude-sonnet-4.5.json index 2721b6a2b6511249d3161ee660ebb342753596b5..66356dbecae8af8177ad511141877805ad8752da 100644 --- a/data/models/anthropic_claude-sonnet-4.5.json +++ b/data/models/anthropic_claude-sonnet-4.5.json @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2026-01-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,11 @@ "max_score": 100.0 }, "score_details": { - "score": 40.1, - "uncertainty": { - "standard_error": { - "value": 2.9 - }, - "num_samples": 435 - } + "score": 42.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +212,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +226,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/maya__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +250,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,11 +259,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.7 + "score": 42.6, + "uncertainty": { + "standard_error": { + "value": 2.8 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"MAYA\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -300,7 +300,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/claude-code__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -324,7 +324,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -333,17 +333,17 @@ "max_score": 100.0 }, "score_details": { - "score": 42.5, + "score": 40.1, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -360,7 +360,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Claude Code\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -448,7 +448,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__claude-sonnet-4.5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__claude-sonnet-4.5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -472,7 +472,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -481,7 +481,7 @@ "max_score": 100.0 }, "score_details": { - "score": 42.6, + "score": 42.5, "uncertainty": { "standard_error": { "value": 2.8 @@ -491,7 +491,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -508,7 +508,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Claude Sonnet 4.5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Claude Sonnet 4.5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json index eb03ad8b35e71a4f26603a2960132d29f0246b60..ed5c0ad9e23e3e1024094c595c2931cbc579e102 100644 --- a/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json +++ b/data/models/boltmonkey_neuraldaredevil-supernova-lite-7b-dareties-abliterated.json @@ -5,7 +5,7 @@ "developer": "BoltMonkey", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7999 + "score": 0.459 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5152 + "score": 0.5185 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1193 + "score": 0.0937 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.281 + "score": 0.2743 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.4083 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3631 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.459 + "score": 0.7999 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5185 + "score": 0.5152 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.1193 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2743 + "score": 0.281 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4083 + "score": 0.4019 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3631 + "score": 0.3733 } } ], diff --git a/data/models/bunnycore_llama-3.2-3b-deep-test.json b/data/models/bunnycore_llama-3.2-3b-deep-test.json index 05cbb770cffa010b799551a630bc7b1089a91900..c829321bc4e1b62229f276602c62e6f6bd9a3faf 100644 --- a/data/models/bunnycore_llama-3.2-3b-deep-test.json +++ b/data/models/bunnycore_llama-3.2-3b-deep-test.json @@ -5,9 +5,9 @@ "developer": "bunnycore", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", - "params_billions": "3.607" + "params_billions": "1.803" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4652 + "score": 0.1775 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4531 + "score": 0.295 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1284 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2643 + "score": 0.2517 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3394 + "score": 0.3647 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3152 + "score": 0.1049 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1775 + "score": 0.4652 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.295 + "score": 0.4531 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1284 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2643 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3647 + "score": 0.3394 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1049 + "score": 0.3152 } } ], diff --git a/data/models/cir-ams_btrm_qwen2_7b_0613.json b/data/models/cir-ams_btrm_qwen2_7b_0613.json index 9fb827ce32fb32044e2247d7f86c70d1bc13d414..84a36ab2262aa5d4869e5d141d383b85699971f1 100644 --- a/data/models/cir-ams_btrm_qwen2_7b_0613.json +++ b/data/models/cir-ams_btrm_qwen2_7b_0613.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.5736 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.8172 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3563 + "score": 0.9749 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.5724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7178 + "score": 0.9014 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5737 + "score": 0.8775 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6527 + "score": 0.7029 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", + "evaluation_id": "reward-bench-2/CIR-AMS_BTRM_Qwen2_7b_0613/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8172 + "score": 0.5736 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9749 + "score": 0.5347 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5724 + "score": 0.3563 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6066 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9014 + "score": 0.7178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8775 + "score": 0.5737 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7029 + "score": 0.6527 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/cohere_command-a-03-2025.json b/data/models/cohere_command-a-03-2025.json index 205f7c496ed81b612dd2b182b3cf2e9fd9ac9c54..aedaf4e5f944a400d87afc543fa422bde24be135 100644 --- a/data/models/cohere_command-a-03-2025.json +++ b/data/models/cohere_command-a-03-2025.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/cohere_command-a-03-2025/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json b/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json index e1500a121f56ad89301900472e223d7937764cb0..fcba7390760a9757eb868fa82619eef91d551f86 100644 --- a/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json +++ b/data/models/columbia-nlp_lion-gemma-2b-dpo-v1.0.json @@ -5,7 +5,7 @@ "developer": "Columbia-NLP", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "GemmaForCausalLM", "params_billions": "2.506" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3278 + "score": 0.3102 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.392 + "score": 0.3881 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0431 + "score": 0.0536 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2492 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.412 + "score": 0.4081 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1666 + "score": 0.1665 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3102 + "score": 0.3278 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3881 + "score": 0.392 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0536 + "score": 0.0431 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2492 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4081 + "score": 0.412 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1665 + "score": 0.1666 } } ], diff --git a/data/models/daemontatox_pathfinderai.json b/data/models/daemontatox_pathfinderai.json index 8b13e2aabf79ff36b82f8d8bd4c0bcdf2b41e385..7a5f7d25c7278e2df08548a48abfe0b0ee9b4f2a 100644 --- a/data/models/daemontatox_pathfinderai.json +++ b/data/models/daemontatox_pathfinderai.json @@ -5,7 +5,7 @@ "developer": "Daemontatox", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "32.764" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3745 + "score": 0.4855 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6668 + "score": 0.6627 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4758 + "score": 0.4841 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3943 + "score": 0.3096 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4858 + "score": 0.4256 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5593 + "score": 0.5542 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4855 + "score": 0.3745 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6627 + "score": 0.6668 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4841 + "score": 0.4758 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3943 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4256 + "score": 0.4858 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5542 + "score": 0.5593 } } ], diff --git a/data/models/davielion_llama-3.2-1b-spin-iter0.json b/data/models/davielion_llama-3.2-1b-spin-iter0.json index 849a1d52d59a5f16eab6aaa35f259f630d9dc175..2ea14b6e6da41d61b06880761473f45de63d9756 100644 --- a/data/models/davielion_llama-3.2-1b-spin-iter0.json +++ b/data/models/davielion_llama-3.2-1b-spin-iter0.json @@ -5,7 +5,7 @@ "developer": "DavieLion", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1549 + "score": 0.1507 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2937 + "score": 0.293 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.006 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2534 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1125 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1507 + "score": 0.1549 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.293 + "score": 0.2937 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.006 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2576 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1125 + "score": 0.1128 } } ], diff --git a/data/models/davielion_llama-3.2-1b-spin-iter3.json b/data/models/davielion_llama-3.2-1b-spin-iter3.json index 99c7b4d0290325891f13fa396abea88044dda26d..7ac3466820dc73e6944e390d07375635e5394c56 100644 --- a/data/models/davielion_llama-3.2-1b-spin-iter3.json +++ b/data/models/davielion_llama-3.2-1b-spin-iter3.json @@ -5,7 +5,7 @@ "developer": "DavieLion", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "1.236" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1324 + "score": 0.1336 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2972 + "score": 0.2975 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0068 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2643 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3527 + "score": 0.35 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1128 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1336 + "score": 0.1324 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2975 + "score": 0.2972 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0068 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2643 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.3527 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1129 } } ], diff --git a/data/models/deepmount00_llama-3.1-8b-ita.json b/data/models/deepmount00_llama-3.1-8b-ita.json index 1fef7ca0b3d471692e7379016093b45312ecc7df..be94466036e753700b8de72e15b468fc6edebdee 100644 --- a/data/models/deepmount00_llama-3.1-8b-ita.json +++ b/data/models/deepmount00_llama-3.1-8b-ita.json @@ -6,8 +6,8 @@ "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", - "architecture": "LlamaForCausalLM", - "params_billions": "8.03", + "architecture": "Unknown", + "params_billions": "0.0", "model_id_aliases": [ "DeepMount00/Llama-3.1-8b-Ita" ] @@ -15,7 +15,7 @@ }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -47,7 +47,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7917 + "score": 0.5365 } }, { @@ -65,7 +65,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5109 + "score": 0.517 } }, { @@ -83,7 +83,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1088 + "score": 0.1707 } }, { @@ -101,7 +101,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2878 + "score": 0.3062 } }, { @@ -119,7 +119,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4136 + "score": 0.4487 } }, { @@ -137,7 +137,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3876 + "score": 0.396 } } ], @@ -145,7 +145,7 @@ "generation_config": null }, { - "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-Ita/1773936498.240187", + "evaluation_id": "hfopenllm_v2/DeepMount00_Llama-3.1-8b-ITA/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -177,7 +177,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5365 + "score": 0.7917 } }, { @@ -195,7 +195,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.517 + "score": 0.5109 } }, { @@ -213,7 +213,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1707 + "score": 0.1088 } }, { @@ -231,7 +231,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.2878 } }, { @@ -249,7 +249,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4487 + "score": 0.4136 } }, { @@ -267,7 +267,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.396 + "score": 0.3876 } } ], diff --git a/data/models/deepseek_deepseek-r1-0528.json b/data/models/deepseek_deepseek-r1-0528.json index 712c8d82096fdfa393951ac4421e5d6c18ee722d..22f7f1e7159f38877f415a9a6f07905fabcb5d33 100644 --- a/data/models/deepseek_deepseek-r1-0528.json +++ b/data/models/deepseek_deepseek-r1-0528.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-r1-0528/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/deepseek_deepseek-v3.1.json b/data/models/deepseek_deepseek-v3.1.json index 52e144fa11c26943b339cba2c9fa28b1d9098c61..28271d97eb42a16e44db35269bdfee0d33d2475a 100644 --- a/data/models/deepseek_deepseek-v3.1.json +++ b/data/models/deepseek_deepseek-v3.1.json @@ -7,8 +7,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -522,8 +522,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/deepseek_deepseek-v3.1/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/dfurman_llama-3-8b-orpo-v0.1.json b/data/models/dfurman_llama-3-8b-orpo-v0.1.json index a2dcf54fc8dfd61d5182710e882025f5ed46cdab..987b3112f03b34626e0421787b1c9cb2e2b55a46 100644 --- a/data/models/dfurman_llama-3-8b-orpo-v0.1.json +++ b/data/models/dfurman_llama-3-8b-orpo-v0.1.json @@ -5,8 +5,8 @@ "developer": "dfurman", "inference_platform": "unknown", "additional_details": { - "precision": "float16", - "architecture": "?", + "precision": "bfloat16", + "architecture": "LlamaForCausalLM", "params_billions": "8.03" } }, @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2835 + "score": 0.3 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.3853 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0521 + "score": 0.0415 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2609 + "score": 0.2617 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3566 + "score": 0.3579 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2298 + "score": 0.2281 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3 + "score": 0.2835 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3853 + "score": 0.3842 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0415 + "score": 0.0521 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2617 + "score": 0.2609 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3579 + "score": 0.3566 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2281 + "score": 0.2298 } } ], diff --git a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json index 623bd9c2c8bfaa513dd516b6dc8551699ef2135b..34df651aabdc1d7685c9419a18517c0c9ee2b5a2 100644 --- a/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json +++ b/data/models/epistemeai_fireball-meta-llama-3.1-8b-instruct-agent-0.003-128k-code-ds-auto.json @@ -5,7 +5,7 @@ "developer": "EpistemeAI", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7207 + "score": 0.7305 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.461 + "score": 0.4649 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1314 + "score": 0.1397 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2701 + "score": 0.2659 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3432 + "score": 0.3209 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3354 + "score": 0.348 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7305 + "score": 0.7207 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4649 + "score": 0.461 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1397 + "score": 0.1314 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.2701 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3209 + "score": 0.3432 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.348 + "score": 0.3354 } } ], diff --git a/data/models/etherll_herplete-llm-llama-3.1-8b.json b/data/models/etherll_herplete-llm-llama-3.1-8b.json index 1f50fb1261d1ec9c67c961ac513392f51c62e5ca..31e2291931d31d5d1deeb312e00cc5de025980fd 100644 --- a/data/models/etherll_herplete-llm-llama-3.1-8b.json +++ b/data/models/etherll_herplete-llm-llama-3.1-8b.json @@ -5,7 +5,7 @@ "developer": "Etherll", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4672 + "score": 0.6106 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5013 + "score": 0.5347 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0279 + "score": 0.1548 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3146 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3991 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3482 + "score": 0.3752 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6106 + "score": 0.4672 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5347 + "score": 0.5013 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1548 + "score": 0.0279 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3146 + "score": 0.2861 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3991 + "score": 0.386 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3752 + "score": 0.3482 } } ], diff --git a/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json b/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json index 57e8dc5e192d9afe5d8658b037afb0f25366b018..4e742d684d75a21064054d5a573ca557e7b1e590 100644 --- a/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json +++ b/data/models/goekdeniz-guelmez_josiefied-qwen2.5-0.5b-instruct-abliterated-v1.json @@ -5,7 +5,7 @@ "developer": "Goekdeniz-Guelmez", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "0.63" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3417 + "score": 0.3472 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3292 + "score": 0.3268 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0023 + "score": 0.0891 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2517 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3249 + "score": 0.3262 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1638 + "score": 0.1641 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3472 + "score": 0.3417 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3268 + "score": 0.3292 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0891 + "score": 0.0023 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2517 + "score": 0.2576 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3262 + "score": 0.3249 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1641 + "score": 0.1638 } } ], diff --git a/data/models/google_gemini-2.5-flash.json b/data/models/google_gemini-2.5-flash.json index a2e65a21393c95ddfc5f4fe71d47a11fac033db8..5eaa67bbe68ea1014fddc37d9abaae6354738cf4 100644 --- a/data/models/google_gemini-2.5-flash.json +++ b/data/models/google_gemini-2.5-flash.json @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 17.1, + "score": 15.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 16.4, + "score": 17.1, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gemini-2.5-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 15.4, + "score": 16.4, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Gemini 2.5 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-2.5-pro.json b/data/models/google_gemini-2.5-pro.json index 64bf8e5b0bbd8066e28305bcfb70c0755706abd7..ad69bc6f71351eff92acf181b72bb1ef3d2ca29a 100644 --- a/data/models/google_gemini-2.5-pro.json +++ b/data/models/google_gemini-2.5-pro.json @@ -1269,7 +1269,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1293,7 +1293,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1302,17 +1302,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.1, + "score": 32.6, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1329,7 +1329,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1343,7 +1343,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1367,7 +1367,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1376,17 +1376,17 @@ "max_score": 100.0 }, "score_details": { - "score": 19.6, + "score": 26.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1403,7 +1403,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1417,7 +1417,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-2.5-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-2.5-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -1441,7 +1441,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -1450,17 +1450,17 @@ "max_score": 100.0 }, "score_details": { - "score": 32.6, + "score": 19.6, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -1477,7 +1477,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 2.5 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 2.5 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini-3-flash.json b/data/models/google_gemini-3-flash.json index 96d81804f9403e5af725c031c3f055444b4c7604..6d70bb6cf77fa0c54cb9eb4fa0ff050319c617e5 100644 --- a/data/models/google_gemini-3-flash.json +++ b/data/models/google_gemini-3-flash.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { - "agent_name": "Gemini CLI", - "agent_organization": "Google" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.0, + "score": 51.7, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-07", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 51.7, + "score": 47.4, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,7 +265,7 @@ "max_score": 100.0 }, "score_details": { - "score": 47.4, + "score": 51.0, "uncertainty": { "standard_error": { "value": 3.0 diff --git a/data/models/google_gemini-3-pro-preview.json b/data/models/google_gemini-3-pro-preview.json index 7edffa00c381bc6c2a68a0a203fbe30120c52254..fdc61e428a001318c9e1c391ace7515f63b19817 100644 --- a/data/models/google_gemini-3-pro-preview.json +++ b/data/models/google_gemini-3-pro-preview.json @@ -4,13 +4,13 @@ "id": "google/gemini-3-pro-preview", "developer": "Google", "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.582, + "score": 0.13, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "8.7", - "total_run_cost": "869.55", - "average_steps": "33.49", - "percent_finished": "0.98" + "average_agent_cost": "2.54", + "total_run_cost": "254.25", + "average_steps": "49.13", + "percent_finished": "0.71" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -91,42 +91,42 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "browsecompplus", + "benchmark": "appworld_test_normal", "evaluation_results": [ { - "evaluation_name": "browsecompplus", + "evaluation_name": "appworld/test_normal", "source_data": { - "dataset_name": "browsecompplus", + "dataset_name": "appworld/test_normal", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "BrowseCompPlus benchmark evaluation", + "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.505, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.85", - "total_run_cost": "284.68", - "average_steps": "22.88", - "percent_finished": "0.7" + "average_agent_cost": "1.88", + "total_run_cost": "188.19", + "average_steps": "21.76", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.505, + "score": 0.582, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.88", - "total_run_cost": "188.19", - "average_steps": "21.76", - "percent_finished": "0.99" + "average_agent_cost": "8.7", + "total_run_cost": "869.55", + "average_steps": "33.49", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -246,23 +246,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.36, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "3.11", - "total_run_cost": "310.55", - "average_steps": "38.01", - "percent_finished": "0.86" + "average_agent_cost": "1.3", + "total_run_cost": "130.49", + "average_steps": "22.59", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -274,15 +274,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -295,34 +295,34 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "appworld_test_normal", + "benchmark": "browsecompplus", "evaluation_results": [ { - "evaluation_name": "appworld/test_normal", + "evaluation_name": "browsecompplus", "source_data": { - "dataset_name": "appworld/test_normal", + "dataset_name": "browsecompplus", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "AppWorld benchmark evaluation (test_normal subset)", + "evaluation_description": "BrowseCompPlus benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.13, + "score": 0.57, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.54", - "total_run_cost": "254.25", - "average_steps": "49.13", - "percent_finished": "0.71" + "average_agent_cost": "2.39", + "total_run_cost": "239.0", + "average_steps": "29.63", + "percent_finished": "0.69" } }, "generation_config": { @@ -350,7 +350,7 @@ } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "appworld/test_normal/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,23 +382,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.36, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "1.3", - "total_run_cost": "130.49", - "average_steps": "22.59", - "percent_finished": "1.0" + "average_agent_cost": "3.11", + "total_run_cost": "310.55", + "average_steps": "38.01", + "percent_finished": "0.86" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "2.39", - "total_run_cost": "239.0", - "average_steps": "29.63", - "percent_finished": "0.69" + "average_agent_cost": "2.85", + "total_run_cost": "284.68", + "average_steps": "22.88", + "percent_finished": "0.7" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -478,15 +478,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -518,23 +518,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.3333, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.44", - "total_run_cost": "44.18", - "average_steps": "7.85", - "percent_finished": "0.99" + "average_agent_cost": "0.64", + "total_run_cost": "63.79", + "average_steps": "8.45", + "percent_finished": "0.6061" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -546,8 +546,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -622,7 +622,7 @@ } }, { - "evaluation_id": "browsecompplus/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -654,23 +654,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3333, + "score": 0.48, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.64", - "total_run_cost": "63.79", - "average_steps": "8.45", - "percent_finished": "0.6061" + "average_agent_cost": "0.44", + "total_run_cost": "44.18", + "average_steps": "7.85", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1720,7 +1720,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1767,8 +1767,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1780,15 +1780,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1820,14 +1820,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7234, + "score": 0.7576, "uncertainty": { - "num_samples": 94 + "num_samples": 99 }, "details": { - "average_agent_cost": "1.58", - "total_run_cost": "148.44", - "average_steps": "32.36", + "average_agent_cost": "2.21", + "total_run_cost": "218.76", + "average_steps": "38.1", "percent_finished": "1.0" } }, @@ -1835,8 +1835,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1848,15 +1848,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1888,14 +1888,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.7234, "uncertainty": { - "num_samples": 99 + "num_samples": 94 }, "details": { - "average_agent_cost": "2.21", - "total_run_cost": "218.76", - "average_steps": "38.1", + "average_agent_cost": "1.58", + "total_run_cost": "148.44", + "average_steps": "32.36", "percent_finished": "1.0" } }, @@ -1903,8 +1903,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1916,8 +1916,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1992,7 +1992,7 @@ } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2039,8 +2039,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2052,15 +2052,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2097,9 +2097,9 @@ "num_samples": 50 }, "details": { - "average_agent_cost": "0.34", - "total_run_cost": "17.45", - "average_steps": "12.62", + "average_agent_cost": "0.16", + "total_run_cost": "8.48", + "average_steps": "10.14", "percent_finished": "1.0" } }, @@ -2107,8 +2107,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2120,15 +2120,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2160,14 +2160,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.62, + "score": 0.7, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.18", - "average_steps": "10.9", + "average_agent_cost": "0.34", + "total_run_cost": "17.45", + "average_steps": "12.62", "percent_finished": "1.0" } }, @@ -2175,8 +2175,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2188,15 +2188,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2228,14 +2228,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7, + "score": 0.62, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.16", - "total_run_cost": "8.48", - "average_steps": "10.14", + "average_agent_cost": "0.21", + "total_run_cost": "11.18", + "average_steps": "10.9", "percent_finished": "1.0" } }, @@ -2243,8 +2243,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -2256,15 +2256,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2311,8 +2311,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -2324,15 +2324,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2345,33 +2345,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7576, + "score": 0.68, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "21.43", - "average_steps": "11.3", + "average_agent_cost": "0.2", + "total_run_cost": "10.29", + "average_steps": "12.28", "percent_finished": "1.0" } }, @@ -2468,7 +2468,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2500,14 +2500,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7805, + "score": 0.7576, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.19", - "total_run_cost": "19.38", - "average_steps": "11.18", + "average_agent_cost": "0.21", + "total_run_cost": "21.43", + "average_steps": "11.3", "percent_finished": "1.0" } }, @@ -2515,8 +2515,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2528,8 +2528,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2672,7 +2672,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2685,33 +2685,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.7805, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.2", - "total_run_cost": "10.29", - "average_steps": "12.28", + "average_agent_cost": "0.19", + "total_run_cost": "19.38", + "average_steps": "11.18", "percent_finished": "1.0" } }, @@ -2719,8 +2719,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2732,15 +2732,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2772,23 +2772,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.8876, + "score": 0.6852, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.54", - "total_run_cost": "58.29", - "average_steps": "10.82", - "percent_finished": "0.89" + "average_agent_cost": "0.21", + "total_run_cost": "25.48", + "average_steps": "9.9", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2800,15 +2800,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2840,14 +2840,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.88, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.35", - "total_run_cost": "40.25", - "average_steps": "12.71", + "average_agent_cost": "0.3", + "total_run_cost": "36.75", + "average_steps": "14.84", "percent_finished": "1.0" } }, @@ -2855,8 +2855,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2868,15 +2868,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2908,14 +2908,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6852, + "score": 0.88, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "25.48", - "average_steps": "9.9", + "average_agent_cost": "0.35", + "total_run_cost": "40.25", + "average_steps": "12.71", "percent_finished": "1.0" } }, @@ -2923,8 +2923,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -2936,15 +2936,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2976,23 +2976,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.8876, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "36.75", - "average_steps": "14.84", - "percent_finished": "1.0" + "average_agent_cost": "0.54", + "total_run_cost": "58.29", + "average_steps": "10.82", + "percent_finished": "0.89" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -3004,15 +3004,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__google_gemini-3-pro-preview/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__google_gemini-3-pro-preview/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -3059,8 +3059,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -3072,8 +3072,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } diff --git a/data/models/google_gemini-3-pro.json b/data/models/google_gemini-3-pro.json index 01e84795b0cee23dd2bc66c9e2c84da542f9aae9..3f400faaadf810745e49d759105088e642b483cc 100644 --- a/data/models/google_gemini-3-pro.json +++ b/data/models/google_gemini-3-pro.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-21", + "evaluation_timestamp": "2026-01-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.9, + "score": 69.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-05", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 62.2, + "score": 61.1, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codebrain-1__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2026-02-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.8, + "score": 62.2, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"CodeBrain-1\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ante__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ii-agent__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-06", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 69.4, + "score": 61.8, "uncertainty": { "standard_error": { - "value": 2.1 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Ante\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"II-Agent\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 61.1, + "score": 56.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -454,7 +454,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gemini-3-pro/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-pro/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -478,7 +478,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-21", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -487,17 +487,17 @@ "max_score": 100.0 }, "score_details": { - "score": 56.0, + "score": 56.9, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -514,7 +514,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"Gemini 3 Pro\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Pro\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/google_gemini_3_pro.json b/data/models/google_gemini_3_pro.json index 8f4f634348fdc10ff492053d2a6241085b80ddd7..104ce8e17e340df7c6fe6dc4bc1f7866f9c1cc71 100644 --- a/data/models/google_gemini_3_pro.json +++ b/data/models/google_gemini_3_pro.json @@ -6,78 +6,6 @@ "inference_platform": "unknown" }, "evaluations": [ - { - "evaluation_id": "ace/google_gemini-3-pro/1773260200", - "retrieved_timestamp": "1773260200", - "source_metadata": { - "source_name": "Mercor ACE Leaderboard", - "source_type": "evaluation_run", - "source_organization_name": "Mercor", - "source_organization_url": "https://www.mercor.com", - "evaluator_relationship": "first_party" - }, - "eval_library": { - "name": "archipelago", - "version": "1.0.0" - }, - "benchmark": "ace", - "evaluation_results": [ - { - "evaluation_name": "Overall Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Overall ACE score (paper snapshot, approximate).", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.47 - }, - "generation_config": { - "additional_details": { - "run_setting": "High", - "value_quality": "approximate" - } - } - }, - { - "evaluation_name": "Gaming Score", - "source_data": { - "dataset_name": "ace", - "source_type": "hf_dataset", - "hf_repo": "Mercor/ACE" - }, - "metric_config": { - "evaluation_description": "Gaming domain score.", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.509 - }, - "generation_config": { - "additional_details": { - "run_setting": "High" - } - } - } - ], - "detailed_evaluation_results": null, - "generation_config": { - "additional_details": { - "run_setting": "High", - "value_quality": "approximate" - } - } - }, { "evaluation_id": "apex-agents/google_gemini-3-pro/1773260200", "retrieved_timestamp": "1773260200", @@ -277,6 +205,78 @@ } } }, + { + "evaluation_id": "ace/google_gemini-3-pro/1773260200", + "retrieved_timestamp": "1773260200", + "source_metadata": { + "source_name": "Mercor ACE Leaderboard", + "source_type": "evaluation_run", + "source_organization_name": "Mercor", + "source_organization_url": "https://www.mercor.com", + "evaluator_relationship": "first_party" + }, + "eval_library": { + "name": "archipelago", + "version": "1.0.0" + }, + "benchmark": "ace", + "evaluation_results": [ + { + "evaluation_name": "Overall Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Overall ACE score (paper snapshot, approximate).", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.47 + }, + "generation_config": { + "additional_details": { + "run_setting": "High", + "value_quality": "approximate" + } + } + }, + { + "evaluation_name": "Gaming Score", + "source_data": { + "dataset_name": "ace", + "source_type": "hf_dataset", + "hf_repo": "Mercor/ACE" + }, + "metric_config": { + "evaluation_description": "Gaming domain score.", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.509 + }, + "generation_config": { + "additional_details": { + "run_setting": "High" + } + } + } + ], + "detailed_evaluation_results": null, + "generation_config": { + "additional_details": { + "run_setting": "High", + "value_quality": "approximate" + } + } + }, { "evaluation_id": "apex-v1/google_gemini-3-pro/1773260200", "retrieved_timestamp": "1773260200", diff --git a/data/models/google_gemma-3-27b-it.json b/data/models/google_gemma-3-27b-it.json index 0d22aa7a55f613493f23d93430a44590b7aa715d..31e90c4548397bec1dec70a558a4830dc0c4f7c9 100644 --- a/data/models/google_gemma-3-27b-it.json +++ b/data/models/google_gemma-3-27b-it.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemma-3-27b-it/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/google_gemma-3-4b-it.json b/data/models/google_gemma-3-4b-it.json index c07b03d8abed5cf353cb3849af3ecfd74770cc92..a4a8d0cbbb50ae74df70e1b97a9540a4077ec070 100644 --- a/data/models/google_gemma-3-4b-it.json +++ b/data/models/google_gemma-3-4b-it.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/google_gemma-3-4b-it/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/google_gemma-3-4b-it/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/google_gemma-3-4b-it/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/google_gemma-3-4b-it/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/guilhermenaturaumana_nature-reason-1.2-reallysmall.json b/data/models/guilhermenaturaumana_nature-reason-1.2-reallysmall.json index 58b844aaea3cd18e9a2e6beb1696adc0f352442e..e6b795b4a42b62e10b5be417dff9c55ab19686de 100644 --- a/data/models/guilhermenaturaumana_nature-reason-1.2-reallysmall.json +++ b/data/models/guilhermenaturaumana_nature-reason-1.2-reallysmall.json @@ -5,7 +5,7 @@ "developer": "GuilhermeNaturaUmana", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4791 + "score": 0.4985 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5649 + "score": 0.5645 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.25 + "score": 0.2576 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2995 + "score": 0.3003 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4439 + "score": 0.4373 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4408 + "score": 0.4429 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4985 + "score": 0.4791 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5645 + "score": 0.5649 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.25 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3003 + "score": 0.2995 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4373 + "score": 0.4439 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4429 + "score": 0.4408 } } ], diff --git a/data/models/huggingfacetb_smollm2-360m-instruct.json b/data/models/huggingfacetb_smollm2-360m-instruct.json index b9a479d91c3b9226f2e42a0a5392df69dc01d9df..a4edda6b87f3cb7e0eeabaee9333bba91d019d12 100644 --- a/data/models/huggingfacetb_smollm2-360m-instruct.json +++ b/data/models/huggingfacetb_smollm2-360m-instruct.json @@ -5,9 +5,9 @@ "developer": "HuggingFaceTB", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", - "params_billions": "0.36" + "params_billions": "0.362" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.083 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3144 + "score": 0.3053 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0151 + "score": 0.0083 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.255 + "score": 0.2651 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3461 + "score": 0.3423 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1117 + "score": 0.1126 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.083 + "score": 0.3842 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3053 + "score": 0.3144 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0083 + "score": 0.0151 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2651 + "score": 0.255 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3423 + "score": 0.3461 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1126 + "score": 0.1117 } } ], diff --git a/data/models/infly_inf-orm-llama3.1-70b.json b/data/models/infly_inf-orm-llama3.1-70b.json index e7947ee940015eb0652da9a52891a9ab47739595..82e76ad6cd43b3a105b801967d5f616a3924844a 100644 --- a/data/models/infly_inf-orm-llama3.1-70b.json +++ b/data/models/infly_inf-orm-llama3.1-70b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7648 + "score": 0.9511 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7411 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9101 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.9365 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9644 + "score": 0.9912 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/infly_INF-ORM-Llama3.1-70B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.903 + "score": 0.7648 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8622 + "score": 0.7411 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/infly_INF-ORM-Llama3.1-70B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9511 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9101 + "score": 0.9644 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9365 + "score": 0.903 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9912 + "score": 0.8622 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/internlm_internlm2-1_8b-reward.json b/data/models/internlm_internlm2-1_8b-reward.json index db02a96edd7d4deec66c9db68ed23c5ecfdc96f9..fdd95af043dae0ccf6943006b6d82bc9a69847ba 100644 --- a/data/models/internlm_internlm2-1_8b-reward.json +++ b/data/models/internlm_internlm2-1_8b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8217 + "score": 0.3902 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9358 + "score": 0.2758 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6623 + "score": 0.3625 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8162 + "score": 0.4426 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8724 + "score": 0.4711 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-1_8b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3902 + "score": 0.596 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2758 + "score": 0.1934 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-1_8b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3625 + "score": 0.8217 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4426 + "score": 0.9358 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4711 + "score": 0.6623 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.596 + "score": 0.8162 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1934 + "score": 0.8724 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/internlm_internlm2-7b-reward.json b/data/models/internlm_internlm2-7b-reward.json index 5907aad77cdf2d42a7b25e1a1a35520112be4497..850f56827380f314c355f2c532b41bbbb69062e8 100644 --- a/data/models/internlm_internlm2-7b-reward.json +++ b/data/models/internlm_internlm2-7b-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/internlm_internlm2-7b-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/internlm_internlm2-7b-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8759 + "score": 0.5335 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9916 + "score": 0.4211 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6952 + "score": 0.4 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8716 + "score": 0.5628 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9453 + "score": 0.5956 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/internlm_internlm2-7b-reward/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5335 + "score": 0.7051 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4211 + "score": 0.5164 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/internlm_internlm2-7b-reward/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4 + "score": 0.8759 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5628 + "score": 0.9916 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5956 + "score": 0.6952 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7051 + "score": 0.8716 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5164 + "score": 0.9453 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json index 1160e36435338576bea29bef2eb39f295798e22d..88cf241ec28671fc887627a1e7bb1af09c15f9cf 100644 --- a/data/models/jaspionjader_kosmos-evaa-fusion-8b.json +++ b/data/models/jaspionjader_kosmos-evaa-fusion-8b.json @@ -5,7 +5,7 @@ "developer": "jaspionjader", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4345 + "score": 0.4418 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5419 + "score": 0.5406 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1292 + "score": 0.1352 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3062 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3854 + "score": 0.386 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4418 + "score": 0.4345 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5406 + "score": 0.5419 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1352 + "score": 0.1292 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3087 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.386 + "score": 0.3854 } } ], diff --git a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json index 4462dbdfa0461cc5c2fd5d7ac79308c4e0a0cd93..7b050770aeb59148493da52d2a21dee32c7d2c89 100644 --- a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json +++ b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_ia.json @@ -5,7 +5,7 @@ "developer": "LeroyDyer", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "7.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3066 + "score": 0.3036 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4577 + "score": 0.4575 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2995 + "score": 0.3012 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4254 + "score": 0.4253 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2318 + "score": 0.2329 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.3066 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4575 + "score": 0.4577 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3012 + "score": 0.2995 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4253 + "score": 0.4254 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2329 + "score": 0.2318 } } ], diff --git a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json index 90656456ac84eb015314278cafdaa288005eb1ba..199a2574ee197c6be6eb6b38849e4c63ab43b085 100644 --- a/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json +++ b/data/models/leroydyer_spydazweb_ai_humanai_012_instruct_xa.json @@ -5,7 +5,7 @@ "developer": "LeroyDyer", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "7.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3579 + "score": 0.3798 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4477 + "score": 0.4483 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0423 + "score": 0.04 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3096 + "score": 0.3129 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4134 + "score": 0.4148 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2376 + "score": 0.2389 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3798 + "score": 0.3579 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4483 + "score": 0.4477 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.04 + "score": 0.0423 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3129 + "score": 0.3096 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4148 + "score": 0.4134 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2389 + "score": 0.2376 } } ], diff --git a/data/models/lxzgordon_urm-llama-3.1-8b.json b/data/models/lxzgordon_urm-llama-3.1-8b.json index 2ce56c90ce0d5fb5788660f5b3f1f1e179701786..7f03035f2bd809fb14be27130e313f5b86a26f9a 100644 --- a/data/models/lxzgordon_urm-llama-3.1-8b.json +++ b/data/models/lxzgordon_urm-llama-3.1-8b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9294 + "score": 0.7394 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.6884 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8816 + "score": 0.45 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9108 + "score": 0.6393 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9698 + "score": 0.9178 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7394 + "score": 0.9758 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6884 + "score": 0.7653 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/LxzGordon_URM-LLaMa-3.1-8B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.45 + "score": 0.9294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6393 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9178 + "score": 0.8816 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9758 + "score": 0.9108 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7653 + "score": 0.9698 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json index 018a3210e6da83ca7e6d3501fd6a7fefdb10449c..2ac3bb48cceaf59e0da754702b99c90edec98998 100644 --- a/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json +++ b/data/models/magpie-align_llama-3-8b-magpie-align-v0.1.json @@ -5,7 +5,7 @@ "developer": "Magpie-Align", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4118 + "score": 0.4027 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4811 + "score": 0.4789 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.034 + "score": 0.0461 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2752 + "score": 0.2768 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3047 + "score": 0.3087 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3006 + "score": 0.3001 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4027 + "score": 0.4118 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.4811 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0461 + "score": 0.034 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2768 + "score": 0.2752 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3087 + "score": 0.3047 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3001 + "score": 0.3006 } } ], diff --git a/data/models/microsoft_phi-3-mini-4k-instruct.json b/data/models/microsoft_phi-3-mini-4k-instruct.json index 9787f21694f92686e734f02c35c722665943d7d4..f0214854e07bd3ad87162726609ad6d31e31396a 100644 --- a/data/models/microsoft_phi-3-mini-4k-instruct.json +++ b/data/models/microsoft_phi-3-mini-4k-instruct.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Phi3ForCausalLM", "params_billions": "3.821" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5613 + "score": 0.5477 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5676 + "score": 0.5491 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1163 + "score": 0.1639 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3196 + "score": 0.3322 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.395 + "score": 0.4284 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3866 + "score": 0.4022 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5477 + "score": 0.5613 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5491 + "score": 0.5676 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1639 + "score": 0.1163 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3322 + "score": 0.3196 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4284 + "score": 0.395 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4022 + "score": 0.3866 } } ], diff --git a/data/models/microsoft_phi-4.json b/data/models/microsoft_phi-4.json index c9fca4565946dbb26825690fda8366b38bda6709..32eb099ab1f2e61ac11a8ea3a80df6bc18e5c656 100644 --- a/data/models/microsoft_phi-4.json +++ b/data/models/microsoft_phi-4.json @@ -5,7 +5,7 @@ "developer": "microsoft", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Phi3ForCausalLM", "params_billions": "14.66" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0488 + "score": 0.0585 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6703 + "score": 0.6691 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2787 + "score": 0.3165 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.401 + "score": 0.406 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5295 + "score": 0.5287 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0585 + "score": 0.0488 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6691 + "score": 0.6703 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3165 + "score": 0.2787 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.406 + "score": 0.401 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5287 + "score": 0.5295 } } ], diff --git a/data/models/minimax_minimax-m2.1.json b/data/models/minimax_minimax-m2.1.json index a5b957c9841bcbc5016fb37de174c988ce273877..33905b0ffaa72b65b1d021a73faac651c1ebfecd 100644 --- a/data/models/minimax_minimax-m2.1.json +++ b/data/models/minimax_minimax-m2.1.json @@ -4,13 +4,13 @@ "id": "minimax/minimax-m2.1", "developer": "MiniMax", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Crux", + "agent_organization": "Roam" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-23", + "evaluation_timestamp": "2025-12-22", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 36.6, "uncertainty": { "standard_error": { "value": 2.9 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__minimax-m2.1/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__minimax-m2.1/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-22", + "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 36.6, + "score": 29.2, "uncertainty": { "standard_error": { "value": 2.9 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"MiniMax M2.1\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"MiniMax M2.1\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/mistralai_mistral-medium-3.json b/data/models/mistralai_mistral-medium-3.json index 3d0755e0237b9760231fee380b173b29d57eab4a..8c2cd1ffe5583f8e08ea6a503148c17defbde8f4 100644 --- a/data/models/mistralai_mistral-medium-3.json +++ b/data/models/mistralai_mistral-medium-3.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-medium-3/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/mistralai_mistral-small-2503.json b/data/models/mistralai_mistral-small-2503.json index 6df0d972b005ae32b060fdc4673d6092e770670f..be5d73de7278abb3c747dbebd44b60d3fa624503 100644 --- a/data/models/mistralai_mistral-small-2503.json +++ b/data/models/mistralai_mistral-small-2503.json @@ -10,8 +10,8 @@ }, "evaluations": [ { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", - "retrieved_timestamp": "1773936583.743359", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", + "retrieved_timestamp": "1773936496.366405", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", @@ -525,8 +525,8 @@ "generation_config": null }, { - "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936496.366405", - "retrieved_timestamp": "1773936496.366405", + "evaluation_id": "global-mmlu-lite/mistralai_mistral-small-2503/1773936583.743359", + "retrieved_timestamp": "1773936583.743359", "source_metadata": { "source_name": "Global MMLU Lite Leaderboard", "source_type": "documentation", diff --git a/data/models/mistralai_mistral-small-instruct-2409.json b/data/models/mistralai_mistral-small-instruct-2409.json index 5aac65e94f46e67333b7cb66dd77ae7f5415d17b..02a04ce41b1de063befb30b7d94d3ad197cffb0e 100644 --- a/data/models/mistralai_mistral-small-instruct-2409.json +++ b/data/models/mistralai_mistral-small-instruct-2409.json @@ -5,9 +5,9 @@ "developer": "mistralai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MistralForCausalLM", - "params_billions": "22.247" + "params_billions": "22.05" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6283 + "score": 0.667 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.583 + "score": 0.5213 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2039 + "score": 0.1435 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3331 + "score": 0.3238 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4063 + "score": 0.3632 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.396 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.667 + "score": 0.6283 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5213 + "score": 0.583 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1435 + "score": 0.2039 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3331 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3632 + "score": 0.4063 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.396 + "score": 0.4099 } } ], diff --git a/data/models/mistralai_mixtral-8x7b-v0.1.json b/data/models/mistralai_mixtral-8x7b-v0.1.json index c3ac844f0072de0a86748b754d6c90570298768c..9d997e3527157e47894ae0f49b424a1634297e78 100644 --- a/data/models/mistralai_mixtral-8x7b-v0.1.json +++ b/data/models/mistralai_mixtral-8x7b-v0.1.json @@ -5,7 +5,7 @@ "developer": "mistralai", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "MixtralForCausalLM", "params_billions": "46.703" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2326 + "score": 0.2415 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5098 + "score": 0.5087 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0937 + "score": 0.102 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3205 + "score": 0.3138 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4413 + "score": 0.4321 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3871 + "score": 0.385 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2415 + "score": 0.2326 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5087 + "score": 0.5098 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.102 + "score": 0.0937 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3205 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4321 + "score": 0.4413 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.385 + "score": 0.3871 } } ], diff --git a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json index 7ef165972eafdeec56d923c82e02fdbbc9479eac..d443de39bb7ed82b00df80190432e583c21fd660 100644 --- a/data/models/mlabonne_neuraldaredevil-8b-abliterated.json +++ b/data/models/mlabonne_neuraldaredevil-8b-abliterated.json @@ -5,7 +5,7 @@ "developer": "mlabonne", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4162 + "score": 0.7561 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5124 + "score": 0.5111 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0853 + "score": 0.0906 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3029 + "score": 0.3062 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.415 + "score": 0.4019 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3802 + "score": 0.3841 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7561 + "score": 0.4162 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5111 + "score": 0.5124 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0906 + "score": 0.0853 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3062 + "score": 0.3029 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4019 + "score": 0.415 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3841 + "score": 0.3802 } } ], diff --git a/data/models/moonshot-ai_kimi-k2-instruct.json b/data/models/moonshot-ai_kimi-k2-instruct.json index 758984500ae56b028445a56fb8562c74a90a3a8f..2cfded0145e5c6821159f45b392f6b86e15c7f49 100644 --- a/data/models/moonshot-ai_kimi-k2-instruct.json +++ b/data/models/moonshot-ai_kimi-k2-instruct.json @@ -4,13 +4,13 @@ "id": "moonshot-ai/kimi-k2-instruct", "developer": "Moonshot AI", "additional_details": { - "agent_name": "OpenHands", - "agent_organization": "OpenHands" + "agent_name": "Terminus 2", + "agent_organization": "Terminal Bench" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 26.7, + "score": 27.8, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__kimi-k2-instruct/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__kimi-k2-instruct/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 27.8, + "score": 26.7, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Kimi K2 Instruct\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"Kimi K2 Instruct\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/multiple_multiple.json b/data/models/multiple_multiple.json index 5fae5f56e342c454584c0143a055e41581a00848..c7a1aa9eb1cb1dfc4d88c122729fc6bae4148158 100644 --- a/data/models/multiple_multiple.json +++ b/data/models/multiple_multiple.json @@ -4,13 +4,13 @@ "id": "multiple/multiple", "developer": "Multiple", "additional_details": { - "agent_name": "Warp", - "agent_organization": "Warp" + "agent_name": "Abacus AI Desktop", + "agent_organization": "Abacus.AI" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-20", + "evaluation_timestamp": "2025-12-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,7 +43,7 @@ "max_score": 100.0 }, "score_details": { - "score": 59.1, + "score": 58.4, "uncertainty": { "standard_error": { "value": 2.8 @@ -53,7 +53,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/warp__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-07", + "evaluation_timestamp": "2025-11-20", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 71.0, + "score": 59.1, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Warp\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/abacus-ai-desktop__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-11", + "evaluation_timestamp": "2026-03-05", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 58.4, + "score": 72.4, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Abacus AI Desktop\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-11-11", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,10 +265,10 @@ "max_score": 100.0 }, "score_details": { - "score": 61.2, + "score": 50.1, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.7 }, "num_samples": 435 } @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-11", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,10 +339,10 @@ "max_score": 100.0 }, "score_details": { - "score": 50.1, + "score": 61.2, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 3.0 }, "num_samples": 435 } @@ -380,7 +380,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/ob-1__multiple/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/junie-cli__multiple/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -404,7 +404,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-05", + "evaluation_timestamp": "2026-03-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -413,17 +413,17 @@ "max_score": 100.0 }, "score_details": { - "score": 72.4, + "score": 71.0, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -440,7 +440,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OB-1\" -m \"Multiple\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Multiple\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/nicolinho_qrm-gemma-2-27b.json b/data/models/nicolinho_qrm-gemma-2-27b.json index 1dea90f885df9d34139a9ef21e55b3dcce1a25fd..98185886d3c230ddcd90456c69a5aeed49795fc5 100644 --- a/data/models/nicolinho_qrm-gemma-2-27b.json +++ b/data/models/nicolinho_qrm-gemma-2-27b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9444 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.7853 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9013 + "score": 0.3719 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.927 + "score": 0.6995 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9826 + "score": 0.9578 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.9535 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7853 + "score": 0.8321 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/nicolinho_QRM-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3719 + "score": 0.9444 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6995 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9578 + "score": 0.9013 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9535 + "score": 0.927 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8321 + "score": 0.9826 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/nicolinho_qrm-llama3.1-8b-v2.json b/data/models/nicolinho_qrm-llama3.1-8b-v2.json index 71e586c5d191f366e8b76150e58f0f9807a69f6b..0df8878cad15f33ea391f78f6a5406e147177591 100644 --- a/data/models/nicolinho_qrm-llama3.1-8b-v2.json +++ b/data/models/nicolinho_qrm-llama3.1-8b-v2.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9314 + "score": 0.7074 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.6653 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8684 + "score": 0.4062 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9257 + "score": 0.612 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9677 + "score": 0.9467 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7074 + "score": 0.8909 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6653 + "score": 0.7234 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/nicolinho_QRM-Llama3.1-8B-v2/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4062 + "score": 0.9314 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.612 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9467 + "score": 0.8684 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8909 + "score": 0.9257 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7234 + "score": 0.9677 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/icefog72_icesakev6rp-7b.json b/data/models/nousresearch_yarn-llama-2-7b-128k.json similarity index 88% rename from data/models/icefog72_icesakev6rp-7b.json rename to data/models/nousresearch_yarn-llama-2-7b-128k.json index f3267e4e350d12bc0c15245dbc590141fe9832d9..031faa4d0ff78244e61988ea0bfc24cefdd224de 100644 --- a/data/models/icefog72_icesakev6rp-7b.json +++ b/data/models/nousresearch_yarn-llama-2-7b-128k.json @@ -1,18 +1,18 @@ { "model_info": { - "name": "IceSakeV6RP-7b", - "id": "icefog72/IceSakeV6RP-7b", - "developer": "icefog72", + "name": "Yarn-Llama-2-7b-128k", + "id": "NousResearch/Yarn-Llama-2-7b-128k", + "developer": "NousResearch", "inference_platform": "unknown", "additional_details": { - "precision": "float16", - "architecture": "MistralForCausalLM", - "params_billions": "7.242" + "precision": "bfloat16", + "architecture": "LlamaForCausalLM", + "params_billions": "7.0" } }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/icefog72_IceSakeV6RP-7b/1773936498.240187", + "evaluation_id": "hfopenllm_v2/NousResearch_Yarn-Llama-2-7b-128k/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5033 + "score": 0.1485 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4976 + "score": 0.3248 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0619 + "score": 0.0151 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2911 + "score": 0.2601 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.42 + "score": 0.3967 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3093 + "score": 0.1791 } } ], diff --git a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json index 486ba5e61208261c68f73d7d2bf88d78b1b36131..f86ab61b521749eb0f8fa030a3c29f5a32d327a1 100644 --- a/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json +++ b/data/models/ontocord_wide_3b_sft_stage1.1-ss1-with_generics_intr_math_stories.no_issue.json @@ -5,7 +5,7 @@ "developer": "ontocord", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "3.759" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1128 + "score": 0.1162 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3171 + "score": 0.3184 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0113 + "score": 0.0076 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2685 + "score": 0.2634 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.3447 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1129 + "score": 0.1124 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1162 + "score": 0.1128 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3184 + "score": 0.3171 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0076 + "score": 0.0113 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2634 + "score": 0.2685 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3447 + "score": 0.346 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1124 + "score": 0.1129 } } ], diff --git a/data/models/openai_gpt-4o-2024-08-06.json b/data/models/openai_gpt-4o-2024-08-06.json index ca15abfce433f9cf3b132cc5fefadfcef719867a..4523783fb76c33e56985cf8706dfcde93c76d009 100644 --- a/data/models/openai_gpt-4o-2024-08-06.json +++ b/data/models/openai_gpt-4o-2024-08-06.json @@ -1900,10 +1900,10 @@ } }, { - "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", + "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -1922,104 +1922,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6493 + "score": 0.8673 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5684 + "score": 0.9609 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.761 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.623 + "score": 0.8811 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8619 + "score": 0.8661 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/openai_gpt-4o-2024-08-06/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.6493 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2028,135 +2052,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7819 + "score": 0.5684 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/openai_gpt-4o-2024-08-06/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8673 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9609 + "score": 0.623 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.761 + "score": 0.8619 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8811 + "score": 0.7293 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8661 + "score": 0.7819 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openai_gpt-4o-mini-2024-07-18.json b/data/models/openai_gpt-4o-mini-2024-07-18.json index 1b3fb4c30102ee1f603e3640bb4c4b14c38b8cac..f38f4e1ddedf08dfff3e3ceeb97aeebb3dcce913 100644 --- a/data/models/openai_gpt-4o-mini-2024-07-18.json +++ b/data/models/openai_gpt-4o-mini-2024-07-18.json @@ -2124,10 +2124,10 @@ } }, { - "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -2146,104 +2146,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5796 + "score": 0.8007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4105 + "score": 0.9497 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.6075 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5191 + "score": 0.8081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8374 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/openai_gpt-4o-mini-2024-07-18/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7414 + "score": 0.5796 }, "source_data": { "dataset_name": "RewardBench 2", @@ -2252,135 +2276,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6962 + "score": 0.4105 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/openai_gpt-4o-mini-2024-07-18/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8007 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9497 + "score": 0.5191 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6075 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8081 + "score": 0.7414 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8374 + "score": 0.6962 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/openai_gpt-5-2025-08-07.json b/data/models/openai_gpt-5-2025-08-07.json index 532bfb57f278c67f06480977b0f49dd924ee82c1..0853492fcc4bbd45f07185718454686350fe1be2 100644 --- a/data/models/openai_gpt-5-2025-08-07.json +++ b/data/models/openai_gpt-5-2025-08-07.json @@ -1264,13 +1264,13 @@ } }, { - "evaluation_id": "livecodebenchpro/gpt-5-2025-08-07/1770683238.099205", - "retrieved_timestamp": "1770683238.099205", + "evaluation_id": "livecodebenchpro/gpt-5-2025-08-07/1760492095.8105888", + "retrieved_timestamp": "1760492095.8105888", "source_metadata": { + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation", - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party" + "source_type": "documentation" }, "eval_library": { "name": "unknown", @@ -1280,62 +1280,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", + "metric_config": { + "evaluation_description": "Pass@1 on Hard Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.04225352112676056 + }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Medium Problems", "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.0423 - } - }, - { - "evaluation_name": "Medium Problems", + "score": 0.4084507042253521 + }, "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Easy Problems", "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Easy Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.4085 - } - }, - { - "evaluation_name": "Easy Problems", + "score": 0.8873239436619719 + }, "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] - }, - "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.9014 } } ], @@ -1343,13 +1343,13 @@ "generation_config": null }, { - "evaluation_id": "livecodebenchpro/gpt-5-2025-08-07/1760492095.8105888", - "retrieved_timestamp": "1760492095.8105888", + "evaluation_id": "livecodebenchpro/gpt-5-2025-08-07/1770683238.099205", + "retrieved_timestamp": "1770683238.099205", "source_metadata": { - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation" + "source_type": "documentation", + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party" }, "eval_library": { "name": "unknown", @@ -1359,62 +1359,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", - "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.04225352112676056 - }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Medium Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Hard Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.4084507042253521 - }, + "score": 0.0423 + } + }, + { + "evaluation_name": "Medium Problems", "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Easy Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.8873239436619719 - }, + "score": 0.4085 + } + }, + { + "evaluation_name": "Easy Problems", "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] + }, + "metric_config": { + "evaluation_description": "Pass@1 on Easy Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.9014 } } ], diff --git a/data/models/openai_gpt-5-mini.json b/data/models/openai_gpt-5-mini.json index 8229a45bea599ef0cf4ec037999fa4d4fdccb65f..c2c25e33d8592b0cf47f06cc9166a5f66ec8f591 100644 --- a/data/models/openai_gpt-5-mini.json +++ b/data/models/openai_gpt-5-mini.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5-mini", "developer": "OpenAI", "additional_details": { - "agent_name": "Codex CLI", - "agent_organization": "OpenAI" + "agent_name": "Mini-SWE-Agent", + "agent_organization": "Princeton" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 31.9, + "score": 22.2, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.6 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 29.2, + "score": 24.0, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-12-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 22.2, + "score": 34.8, "uncertainty": { "standard_error": { - "value": 2.6 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/spoox-m__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-24", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 34.8, + "score": 29.2, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"spoox-m\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-mini/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-mini/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 24.0, + "score": 31.9, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Mini\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Mini\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5-nano.json b/data/models/openai_gpt-5-nano.json index 9816cdcd5d573c1531a9f756bbbe26744d018229..f1c6d8f032d8015d1380b26ffc2082a7ad57137f 100644 --- a/data/models/openai_gpt-5-nano.json +++ b/data/models/openai_gpt-5-nano.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,7 +117,7 @@ "max_score": 100.0 }, "score_details": { - "score": 7.0, + "score": 7.9, "uncertainty": { "standard_error": { "value": 1.9 @@ -127,7 +127,7 @@ }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-04", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 7.9, + "score": 11.5, "uncertainty": { "standard_error": { - "value": 1.9 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -232,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5-nano/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5-nano/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -256,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-04", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -265,17 +265,17 @@ "max_score": 100.0 }, "score_details": { - "score": 11.5, + "score": 7.0, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 1.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -292,7 +292,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5-Nano\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5-Nano\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.1-codex.json b/data/models/openai_gpt-5.1-codex.json index a2aac5e629083b81e1729673348d7d31d575cd9b..5ac777453649e155d2cf4ca5c0edb4c1fb8d16a4 100644 --- a/data/models/openai_gpt-5.1-codex.json +++ b/data/models/openai_gpt-5.1-codex.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-17", + "evaluation_timestamp": "2025-11-16", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 53.5, + "score": 57.8, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__gpt-5.1-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/letta-code__gpt-5.1-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-16", + "evaluation_timestamp": "2025-12-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 57.8, + "score": 53.5, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GPT-5.1-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Letta Code\" -m \"GPT-5.1-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.2-2025-12-11.json b/data/models/openai_gpt-5.2-2025-12-11.json index b70c28d6ee613e835fe13ebe0ae4b3e2cc5d0ae8..c3a672d6300dfa51d07ee396880a48a053600e4c 100644 --- a/data/models/openai_gpt-5.2-2025-12-11.json +++ b/data/models/openai_gpt-5.2-2025-12-11.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5.2-2025-12-11", "developer": "OpenAI", "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } }, "evaluations": [ { - "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -42,23 +42,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.071, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.55", - "total_run_cost": "55.03", - "average_steps": "51.59", - "percent_finished": "0.61" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -70,15 +70,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "appworld/test_normal/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -110,23 +110,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.071, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.55", + "total_run_cost": "55.03", + "average_steps": "51.59", + "percent_finished": "0.61" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -138,15 +138,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -178,23 +178,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0, + "score": 0.22, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.0", - "total_run_cost": "0.0", - "average_steps": "0.0", - "percent_finished": "0.0" + "average_agent_cost": "0.36", + "total_run_cost": "36.37", + "average_steps": "10.05", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -206,15 +206,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "appworld/test_normal/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "appworld/test_normal/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -246,23 +246,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.22, + "score": 0.0, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.36", - "total_run_cost": "36.37", - "average_steps": "10.05", - "percent_finished": "1.0" + "average_agent_cost": "0.0", + "total_run_cost": "0.0", + "average_steps": "0.0", + "percent_finished": "0.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -274,8 +274,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -350,7 +350,7 @@ } }, { - "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -382,14 +382,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.43, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.38", - "total_run_cost": "38.21", - "average_steps": "14.27", + "average_agent_cost": "0.43", + "total_run_cost": "43.11", + "average_steps": "8.97", "percent_finished": "1.0" } }, @@ -397,8 +397,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -410,15 +410,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -450,23 +450,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.46, + "score": 0.48, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "29.78", - "average_steps": "8.14", - "percent_finished": "0.99" + "average_agent_cost": "0.38", + "total_run_cost": "38.21", + "average_steps": "14.27", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -478,8 +478,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -554,7 +554,7 @@ } }, { - "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -586,14 +586,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.26, + "score": 0.46, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.17", - "total_run_cost": "17.31", - "average_steps": "6.57", + "average_agent_cost": "0.3", + "total_run_cost": "29.78", + "average_steps": "8.14", "percent_finished": "0.99" } }, @@ -601,8 +601,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -614,15 +614,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "browsecompplus/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "browsecompplus/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -654,23 +654,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.43, + "score": 0.26, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.43", - "total_run_cost": "43.11", - "average_steps": "8.97", - "percent_finished": "1.0" + "average_agent_cost": "0.17", + "total_run_cost": "17.31", + "average_steps": "6.57", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -682,8 +682,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -769,7 +769,7 @@ "generation_config": null }, { - "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -801,14 +801,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.58, + "score": 0.57, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.94", - "total_run_cost": "93.98", - "average_steps": "23.99", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -816,8 +816,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -829,15 +829,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -869,14 +869,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5455, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.26", + "total_run_cost": "25.64", + "average_steps": "20.44", "percent_finished": "1.0" } }, @@ -884,8 +884,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -897,15 +897,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "swe-bench/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -937,14 +937,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.57, + "score": 0.5253, "uncertainty": { - "num_samples": 100 + "num_samples": 99 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "24.76", - "average_steps": "20.47", + "average_agent_cost": "0.45", + "total_run_cost": "44.58", + "average_steps": "19.98", "percent_finished": "1.0" } }, @@ -952,8 +952,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -965,15 +965,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "swe-bench/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1005,14 +1005,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5253, + "score": 0.58, "uncertainty": { - "num_samples": 99 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.45", - "total_run_cost": "44.58", - "average_steps": "19.98", + "average_agent_cost": "0.94", + "total_run_cost": "93.98", + "average_steps": "23.99", "percent_finished": "1.0" } }, @@ -1020,8 +1020,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1033,15 +1033,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "swe-bench/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1054,33 +1054,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "swe-bench", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "swe-bench", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "swe-bench", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "SWE-bench benchmark evaluation", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5455, + "score": 0.5, "uncertainty": { - "num_samples": 99 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.26", - "total_run_cost": "25.64", - "average_steps": "20.44", + "average_agent_cost": "0.11", + "total_run_cost": "5.77", + "average_steps": "11.4", "percent_finished": "1.0" } }, @@ -1109,7 +1109,7 @@ } }, { - "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "swe-bench/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1122,33 +1122,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "swe-bench", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "swe-bench", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "swe-bench", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "SWE-bench benchmark evaluation", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.48, + "score": 0.57, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.21", - "total_run_cost": "11.23", - "average_steps": "10.18", + "average_agent_cost": "0.25", + "total_run_cost": "24.76", + "average_steps": "20.47", "percent_finished": "1.0" } }, @@ -1156,8 +1156,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1169,15 +1169,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1209,14 +1209,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5, + "score": 0.54, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "5.77", - "average_steps": "11.4", + "average_agent_cost": "0.13", + "total_run_cost": "6.96", + "average_steps": "11.22", "percent_finished": "1.0" } }, @@ -1224,8 +1224,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1237,15 +1237,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1277,14 +1277,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.54, + "score": 0.48, "uncertainty": { "num_samples": 50 }, "details": { - "average_agent_cost": "0.13", - "total_run_cost": "6.96", - "average_steps": "11.22", + "average_agent_cost": "0.21", + "total_run_cost": "11.23", + "average_steps": "10.18", "percent_finished": "1.0" } }, @@ -1292,8 +1292,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1305,15 +1305,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/airline/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1360,8 +1360,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1373,15 +1373,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1394,33 +1394,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_retail", + "benchmark": "tau-bench-2_airline", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/retail", + "evaluation_name": "tau-bench-2/airline", "source_data": { - "dataset_name": "tau-bench-2/retail", + "dataset_name": "tau-bench-2/airline", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.68, + "score": 0.6, "uncertainty": { - "num_samples": 100 + "num_samples": 50 }, "details": { - "average_agent_cost": "0.25", - "total_run_cost": "26.27", - "average_steps": "11.08", + "average_agent_cost": "0.29", + "total_run_cost": "15.28", + "average_steps": "10.68", "percent_finished": "1.0" } }, @@ -1449,7 +1449,7 @@ } }, { - "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1481,23 +1481,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.51, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.12", - "total_run_cost": "12.63", - "average_steps": "9.92", - "percent_finished": "0.98" + "average_agent_cost": "0.11", + "total_run_cost": "11.54", + "average_steps": "9.55", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1509,15 +1509,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/airline/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1530,33 +1530,33 @@ "name": "exgentic", "version": "0.1.0" }, - "benchmark": "tau-bench-2_airline", + "benchmark": "tau-bench-2_retail", "evaluation_results": [ { - "evaluation_name": "tau-bench-2/airline", + "evaluation_name": "tau-bench-2/retail", "source_data": { - "dataset_name": "tau-bench-2/airline", + "dataset_name": "tau-bench-2/retail", "source_type": "url", "url": [ "https://github.com/Exgentic/exgentic" ] }, "metric_config": { - "evaluation_description": "Tau Bench 2 benchmark evaluation (airline subset)", + "evaluation_description": "Tau Bench 2 benchmark evaluation (retail subset)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6, + "score": 0.73, "uncertainty": { - "num_samples": 50 + "num_samples": 100 }, "details": { - "average_agent_cost": "0.29", - "total_run_cost": "15.28", - "average_steps": "10.68", + "average_agent_cost": "0.11", + "total_run_cost": "12.27", + "average_steps": "10.33", "percent_finished": "1.0" } }, @@ -1564,8 +1564,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1577,15 +1577,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/retail/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1617,23 +1617,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.73, "uncertainty": { "num_samples": 100 }, "details": { "average_agent_cost": "0.11", - "total_run_cost": "11.54", - "average_steps": "9.55", - "percent_finished": "0.99" + "total_run_cost": "12.27", + "average_steps": "10.33", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -1645,15 +1645,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1685,23 +1685,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.51, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", - "percent_finished": "1.0" + "average_agent_cost": "0.12", + "total_run_cost": "12.63", + "average_steps": "9.92", + "percent_finished": "0.98" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -1713,15 +1713,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/retail/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/retail/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1753,14 +1753,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.73, + "score": 0.68, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.11", - "total_run_cost": "12.27", - "average_steps": "10.33", + "average_agent_cost": "0.25", + "total_run_cost": "26.27", + "average_steps": "11.08", "percent_finished": "1.0" } }, @@ -1768,8 +1768,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1781,15 +1781,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1821,23 +1821,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.53, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.15", - "total_run_cost": "18.88", - "average_steps": "9.92", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } @@ -1849,15 +1849,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "OpenAI Solo", - "agent_framework": "openai_solo" + "agent_name": "LiteLLM Tool Calling with Shortlisting", + "agent_framework": "tool_calling_with_shortlisting" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1889,14 +1889,14 @@ "max_score": 1.0 }, "score_details": { - "score": 0.55, + "score": 0.71, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.1", - "total_run_cost": "15.15", - "average_steps": "9.36", + "average_agent_cost": "0.3", + "total_run_cost": "35.31", + "average_steps": "10.11", "percent_finished": "1.0" } }, @@ -1904,8 +1904,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } @@ -1917,15 +1917,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "Claude Code CLI", - "agent_framework": "claude_code" + "agent_name": "SmolAgents Code", + "agent_framework": "smolagents_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling-with-shortlisting__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/openai-solo__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -1957,23 +1957,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.53, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.15", + "total_run_cost": "18.88", + "average_steps": "9.92", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } @@ -1985,15 +1985,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling with Shortlisting", - "agent_framework": "tool_calling_with_shortlisting" + "agent_name": "OpenAI Solo", + "agent_framework": "openai_solo" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/claude-code-cli__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2025,23 +2025,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5354, + "score": 0.55, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.14", - "total_run_cost": "19.92", - "average_steps": "10.18", - "percent_finished": "0.99" + "average_agent_cost": "0.1", + "total_run_cost": "15.15", + "average_steps": "9.36", + "percent_finished": "1.0" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } @@ -2053,15 +2053,15 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "LiteLLM Tool Calling", - "agent_framework": "tool_calling" + "agent_name": "Claude Code CLI", + "agent_framework": "claude_code" } } } } }, { - "evaluation_id": "tau-bench-2/telecom/smolagents-code__openai_gpt-5.2-2025-12-11/1774263615.0201504", + "evaluation_id": "tau-bench-2/telecom/litellm-tool-calling__openai_gpt-5.2-2025-12-11/1774263615.0201504", "retrieved_timestamp": "1774263615.0201504", "source_metadata": { "source_name": "Exgentic Open Agent Leaderboard", @@ -2093,23 +2093,23 @@ "max_score": 1.0 }, "score_details": { - "score": 0.71, + "score": 0.5354, "uncertainty": { "num_samples": 100 }, "details": { - "average_agent_cost": "0.3", - "total_run_cost": "35.31", - "average_steps": "10.11", - "percent_finished": "1.0" + "average_agent_cost": "0.14", + "total_run_cost": "19.92", + "average_steps": "10.18", + "percent_finished": "0.99" } }, "generation_config": { "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } @@ -2121,8 +2121,8 @@ "generation_args": { "agentic_eval_config": { "additional_details": { - "agent_name": "SmolAgents Code", - "agent_framework": "smolagents_code" + "agent_name": "LiteLLM Tool Calling", + "agent_framework": "tool_calling" } } } diff --git a/data/models/openai_gpt-5.2.json b/data/models/openai_gpt-5.2.json index 26fe4dd6a4fc6668b751e0d5655642084b59ba75..1f373630f767c55e79d63767d1456fe1b49fa7e4 100644 --- a/data/models/openai_gpt-5.2.json +++ b/data/models/openai_gpt-5.2.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-12", + "evaluation_timestamp": "2025-12-18", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 54.0, + "score": 62.9, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-17", + "evaluation_timestamp": "2025-12-12", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,11 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 60.7 + "score": 54.0, + "uncertainty": { + "standard_error": { + "value": 2.9 + }, + "num_samples": 435 + } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -212,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -226,7 +232,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/codex-cli__gpt-5.2/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.2/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -250,7 +256,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-12-18", + "evaluation_timestamp": "2026-01-17", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -259,17 +265,11 @@ "max_score": 100.0 }, "score_details": { - "score": 62.9, - "uncertainty": { - "standard_error": { - "value": 3.0 - }, - "num_samples": 435 - } + "score": 60.7 }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -286,7 +286,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Codex CLI\" -m \"GPT-5.2\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.2\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.3-codex.json b/data/models/openai_gpt-5.3-codex.json index d48b8e986526c7f815b5efdb141e51c853537ca1..71af5af4fde2b8fad2d4e8ea66842bf1cb778d6d 100644 --- a/data/models/openai_gpt-5.3-codex.json +++ b/data/models/openai_gpt-5.3-codex.json @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-03-06", + "evaluation_timestamp": "2026-02-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 74.6, + "score": 75.1, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.4 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/simple-codex__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-06", + "evaluation_timestamp": "2026-02-24", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 75.1, + "score": 77.3, "uncertainty": { "standard_error": { - "value": 2.4 + "value": 2.2 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Simple Codex\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -306,7 +306,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/droid__gpt-5.3-codex/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mux__gpt-5.3-codex/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -330,7 +330,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-24", + "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -339,17 +339,17 @@ "max_score": 100.0 }, "score_details": { - "score": 77.3, + "score": 74.6, "uncertainty": { "standard_error": { - "value": 2.2 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -366,7 +366,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Droid\" -m \"GPT-5.3-Codex\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mux\" -m \"GPT-5.3-Codex\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-5.json b/data/models/openai_gpt-5.json index 539597ca005b6b7d3304665ed895ff7bb0860332..9d09df6c74b83cb2a4dd70946717742ac0b7c9f5 100644 --- a/data/models/openai_gpt-5.json +++ b/data/models/openai_gpt-5.json @@ -4,13 +4,13 @@ "id": "openai/gpt-5", "developer": "OpenAI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "OpenHands", + "agent_organization": "OpenHands" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/openhands__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-10-31", + "evaluation_timestamp": "2025-11-02", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 35.2, + "score": 43.8, "uncertainty": { "standard_error": { - "value": 3.1 + "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-10-31", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.9, + "score": 35.2, "uncertainty": { "standard_error": { - "value": 2.9 + "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -158,7 +158,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/openhands__gpt-5/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-5/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -182,7 +182,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-02", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -191,17 +191,17 @@ "max_score": 100.0 }, "score_details": { - "score": 43.8, + "score": 33.9, "uncertainty": { "standard_error": { - "value": 3.0 + "value": 2.9 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -218,7 +218,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"OpenHands\" -m \"GPT-5\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-5\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_gpt-oss-120b.json b/data/models/openai_gpt-oss-120b.json index cc688cac8bf3c3ea698efc0de68329e08bff64c1..396f861a54b2b1a7762f9a8f0810f5cd018d5e52 100644 --- a/data/models/openai_gpt-oss-120b.json +++ b/data/models/openai_gpt-oss-120b.json @@ -310,7 +310,7 @@ "generation_config": null }, { - "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-120b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-120b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -334,7 +334,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-03", + "evaluation_timestamp": "2025-11-01", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -343,17 +343,17 @@ "max_score": 100.0 }, "score_details": { - "score": 14.2, + "score": 18.7, "uncertainty": { "standard_error": { - "value": 2.3 + "value": 2.7 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -370,7 +370,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -384,7 +384,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/terminus-2__gpt-oss-120b/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/mini-swe-agent__gpt-oss-120b/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -408,7 +408,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2025-11-01", + "evaluation_timestamp": "2025-11-03", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -417,17 +417,17 @@ "max_score": 100.0 }, "score_details": { - "score": 18.7, + "score": 14.2, "uncertainty": { "standard_error": { - "value": 2.7 + "value": 2.3 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -444,7 +444,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GPT-OSS-120B\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Mini-SWE-Agent\" -m \"GPT-OSS-120B\" -k 5", "agentic_eval_config": { "available_tools": [ { diff --git a/data/models/openai_o4-mini-2025-04-16.json b/data/models/openai_o4-mini-2025-04-16.json index 2445a24c2417bbf08441a2388a2fba31fa9d6f49..751e567f80107ae7403958c0915864809e731407 100644 --- a/data/models/openai_o4-mini-2025-04-16.json +++ b/data/models/openai_o4-mini-2025-04-16.json @@ -749,13 +749,13 @@ } }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", - "retrieved_timestamp": "1770683238.099205", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", + "retrieved_timestamp": "1760492095.8105888", "source_metadata": { + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation", - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party" + "source_type": "documentation" }, "eval_library": { "name": "unknown", @@ -765,62 +765,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", + "metric_config": { + "evaluation_description": "Pass@1 on Hard Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0, + "max_score": 1 + }, + "score_details": { + "score": 0.014084507042253521 + }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Medium Problems", "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.0143 - } - }, - { - "evaluation_name": "Medium Problems", + "score": 0.30985915492957744 + }, "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - }, + } + }, + { + "evaluation_name": "Easy Problems", "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Easy Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 + "min_score": 0, + "max_score": 1 }, "score_details": { - "score": 0.2923 - } - }, - { - "evaluation_name": "Easy Problems", + "score": 0.8873239436619719 + }, "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] - }, - "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.8571 } } ], @@ -828,13 +828,13 @@ "generation_config": null }, { - "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1760492095.8105888", - "retrieved_timestamp": "1760492095.8105888", + "evaluation_id": "livecodebenchpro/o4-mini-2025-04-16/1770683238.099205", + "retrieved_timestamp": "1770683238.099205", "source_metadata": { - "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", - "evaluator_relationship": "third_party", "source_name": "Live Code Bench Pro", - "source_type": "documentation" + "source_type": "documentation", + "source_organization_name": "New York University, Princeton University, University of California San Diego, University of Washington and Canyon Crest Academy", + "evaluator_relationship": "third_party" }, "eval_library": { "name": "unknown", @@ -844,62 +844,62 @@ "evaluation_results": [ { "evaluation_name": "Hard Problems", - "metric_config": { - "evaluation_description": "Pass@1 on Hard Problems", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0, - "max_score": 1 - }, - "score_details": { - "score": 0.014084507042253521 - }, "source_data": { "dataset_name": "Hard Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=hard&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Medium Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Medium Problems", + "evaluation_description": "Pass@1 on Hard Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.30985915492957744 - }, + "score": 0.0143 + } + }, + { + "evaluation_name": "Medium Problems", "source_data": { "dataset_name": "Medium Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=medium&benchmark_mode=live" ] - } - }, - { - "evaluation_name": "Easy Problems", + }, "metric_config": { - "evaluation_description": "Pass@1 on Easy Problems", + "evaluation_description": "Pass@1 on Medium Problems", "lower_is_better": false, "score_type": "continuous", - "min_score": 0, - "max_score": 1 + "min_score": 0.0, + "max_score": 1.0 }, "score_details": { - "score": 0.8873239436619719 - }, + "score": 0.2923 + } + }, + { + "evaluation_name": "Easy Problems", "source_data": { "dataset_name": "Easy Problems", "source_type": "url", "url": [ "https://webhook.cp-bench.orzzh.com/leaderboard/llm/difficulty?difficulty=easy&benchmark_mode=live" ] + }, + "metric_config": { + "evaluation_description": "Pass@1 on Easy Problems", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.8571 } } ], diff --git a/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json b/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json index 34afb9663c74112d98337da6fd83b3594c465f50..cdfb6729d8ee2c00438e79bd148b5b9825fbce54 100644 --- a/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json +++ b/data/models/openassistant_oasst-rm-2-pythia-6.9b-epoch-1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", + "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.615 + "score": 0.2653 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9246 + "score": 0.3979 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.2875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.377 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5446 + "score": 0.3289 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5855 + "score": 0.1535 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6801 + "score": 0.047 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", + "evaluation_id": "reward-bench/OpenAssistant_oasst-rm-2-pythia-6.9b-epoch-1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2653 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3979 + "score": 0.615 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2875 + "score": 0.9246 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.377 + "score": 0.3728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.5446 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1535 + "score": 0.5855 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.047 + "score": 0.6801 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/openbmb_ultrarm-13b.json b/data/models/openbmb_ultrarm-13b.json index 84bdd483e26f91976f0250093c6ea14b4f0ff97c..c52a509adb327ccd9798d5a844c78601940ebb17 100644 --- a/data/models/openbmb_ultrarm-13b.json +++ b/data/models/openbmb_ultrarm-13b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6903 + "score": 0.4683 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.5063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5548 + "score": 0.3312 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5519 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5986 + "score": 0.5089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6244 + "score": 0.6081 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7294 + "score": 0.3036 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/openbmb_UltraRM-13b/1766412838.146816", + "evaluation_id": "reward-bench/openbmb_UltraRM-13b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4683 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5063 + "score": 0.6903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3312 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5519 + "score": 0.5548 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5089 + "score": 0.5986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6081 + "score": 0.6244 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3036 + "score": 0.7294 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v1.0-cost.json b/data/models/pku-alignment_beaver-7b-v1.0-cost.json index 3777eba3edfdc470c669a503ac85994bf8139135..8e786484059e3101c1249c1cce8c4b2be81faa5a 100644 --- a/data/models/pku-alignment_beaver-7b-v1.0-cost.json +++ b/data/models/pku-alignment_beaver-7b-v1.0-cost.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3332 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3263 + "score": 0.5798 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2313 + "score": 0.6173 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3989 + "score": 0.4232 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7589 + "score": 0.7351 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2939 + "score": 0.5482 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.57 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-cost/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5798 + "score": 0.3332 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6173 + "score": 0.3263 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4232 + "score": 0.2313 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3989 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7351 + "score": 0.7589 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5482 + "score": 0.2939 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.57 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v1.0-reward.json b/data/models/pku-alignment_beaver-7b-v1.0-reward.json index ee66fd95461643d78856676dc7b45eb953fc109f..adf890d6a20deeb5311651870c4e5638866b9733 100644 --- a/data/models/pku-alignment_beaver-7b-v1.0-reward.json +++ b/data/models/pku-alignment_beaver-7b-v1.0-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4727 + "score": 0.1606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.2105 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2873 + "score": 0.2938 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.2623 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3757 + "score": 0.1422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.346 + "score": 0.0646 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5993 + "score": -0.01 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v1.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.1606 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2105 + "score": 0.4727 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2938 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2623 + "score": 0.2873 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1422 + "score": 0.3757 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0646 + "score": 0.346 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": -0.01 + "score": 0.5993 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/pku-alignment_beaver-7b-v2.0-reward.json b/data/models/pku-alignment_beaver-7b-v2.0-reward.json index 4c7c399de00819c04f2b08eddfc902f45548a34f..e6954fa8026e0ae406f243f530096227275cb53f 100644 --- a/data/models/pku-alignment_beaver-7b-v2.0-reward.json +++ b/data/models/pku-alignment_beaver-7b-v2.0-reward.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v2.0-reward/1766412838.146816", + "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v2.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,109 +31,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6366 + "score": 0.2544 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8994 + "score": 0.2168 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.364 + "score": 0.2562 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3825 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6041 + "score": 0.3156 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6887 + "score": 0.2606 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6171 + "score": 0.0944 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], @@ -141,10 +159,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench-2/PKU-Alignment_beaver-7b-v2.0-reward/1766412838.146816", + "evaluation_id": "reward-bench/PKU-Alignment_beaver-7b-v2.0-reward/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -163,127 +181,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2544 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2168 + "score": 0.6366 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2562 + "score": 0.8994 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3825 + "score": 0.364 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3156 + "score": 0.6041 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2606 + "score": 0.6887 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0944 + "score": 0.6171 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/primeintellect_intellect-1.json b/data/models/primeintellect_intellect-1.json index b8fa0710a43fde559062289de677b131f6527599..7d9ec915394b34f60309abc9ed073a5aa8bce5ab 100644 --- a/data/models/primeintellect_intellect-1.json +++ b/data/models/primeintellect_intellect-1.json @@ -5,7 +5,7 @@ "developer": "PrimeIntellect", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "10.211" } @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.274 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.25 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3339 + "score": 0.3753 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1123 + "score": 0.112 } } ], @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.274 + "score": 0.276 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.25 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3753 + "score": 0.3339 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.112 + "score": 0.1123 } } ], diff --git a/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json b/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json index b88fd125d78454888c7b70c3de2f72cbe2fd7732..e065032f9e09241e38d49be10f5ffa3d463b4775 100644 --- a/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json +++ b/data/models/princeton-nlp_llama-3-8b-prolong-512k-instruct.json @@ -140,136 +140,6 @@ ], "detailed_evaluation_results": null, "generation_config": null - }, - { - "evaluation_id": "hfopenllm_v2/princeton-nlp_Llama-3-8B-ProLong-512k-Instruct/1773936498.240187", - "retrieved_timestamp": "1773936498.240187", - "source_metadata": { - "source_name": "HF Open LLM v2", - "source_type": "documentation", - "source_organization_name": "Hugging Face", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "lm-evaluation-harness", - "version": "0.4.0", - "additional_details": { - "fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess" - } - }, - "benchmark": "hfopenllm_v2", - "evaluation_results": [ - { - "evaluation_name": "IFEval", - "source_data": { - "dataset_name": "IFEval", - "source_type": "hf_dataset", - "hf_repo": "google/IFEval" - }, - "metric_config": { - "evaluation_description": "Accuracy on IFEval", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3978 - } - }, - { - "evaluation_name": "BBH", - "source_data": { - "dataset_name": "BBH", - "source_type": "hf_dataset", - "hf_repo": "SaylorTwift/bbh" - }, - "metric_config": { - "evaluation_description": "Accuracy on BBH", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.4983 - } - }, - { - "evaluation_name": "MATH Level 5", - "source_data": { - "dataset_name": "MATH Level 5", - "source_type": "hf_dataset", - "hf_repo": "DigitalLearningGmbH/MATH-lighteval" - }, - "metric_config": { - "evaluation_description": "Exact Match on MATH Level 5", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.0582 - } - }, - { - "evaluation_name": "GPQA", - "source_data": { - "dataset_name": "GPQA", - "source_type": "hf_dataset", - "hf_repo": "Idavidrein/gpqa" - }, - "metric_config": { - "evaluation_description": "Accuracy on GPQA", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.281 - } - }, - { - "evaluation_name": "MUSR", - "source_data": { - "dataset_name": "MUSR", - "source_type": "hf_dataset", - "hf_repo": "TAUR-Lab/MuSR" - }, - "metric_config": { - "evaluation_description": "Accuracy on MUSR", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.425 - } - }, - { - "evaluation_name": "MMLU-PRO", - "source_data": { - "dataset_name": "MMLU-PRO", - "source_type": "hf_dataset", - "hf_repo": "TIGER-Lab/MMLU-Pro" - }, - "metric_config": { - "evaluation_description": "Accuracy on MMLU-PRO", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3246 - } - } - ], - "detailed_evaluation_results": null, - "generation_config": null } ] } \ No newline at end of file diff --git a/data/models/prithivmlmods_calcium-opus-14b-elite.json b/data/models/prithivmlmods_calcium-opus-14b-elite.json index 746fd41957e7795e0f0f3751181ce9bfef3d8e70..89a0dd9277acb0b40655f39809af8ab4d96076ef 100644 --- a/data/models/prithivmlmods_calcium-opus-14b-elite.json +++ b/data/models/prithivmlmods_calcium-opus-14b-elite.json @@ -5,7 +5,7 @@ "developer": "prithivMLmods", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.766" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6064 + "score": 0.6052 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6296 + "score": 0.6317 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3708 + "score": 0.4789 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3733 + "score": 0.3742 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4873 + "score": 0.486 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5307 + "score": 0.5302 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6052 + "score": 0.6064 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6317 + "score": 0.6296 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4789 + "score": 0.3708 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3742 + "score": 0.3733 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.486 + "score": 0.4873 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5302 + "score": 0.5307 } } ], diff --git a/data/models/qingy2019_qwen2.5-math-14b-instruct.json b/data/models/qingy2019_qwen2.5-math-14b-instruct.json index d07b0024c79975f453423bba6d277d902ff3056c..21a12461d0cce8974037575db54605a0da19b661 100644 --- a/data/models/qingy2019_qwen2.5-math-14b-instruct.json +++ b/data/models/qingy2019_qwen2.5-math-14b-instruct.json @@ -5,7 +5,7 @@ "developer": "qingy2019", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6066 + "score": 0.6005 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.635 + "score": 0.6356 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3716 + "score": 0.2764 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3725 + "score": 0.3691 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5331 + "score": 0.5339 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6005 + "score": 0.6066 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6356 + "score": 0.635 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2764 + "score": 0.3716 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3691 + "score": 0.3725 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5339 + "score": 0.5331 } } ], diff --git a/data/models/quazim0t0_casa-14b-sce.json b/data/models/quazim0t0_casa-14b-sce.json index 5121da283045b50c1becfc23584ab4044d6d9721..04555b72d4f62f43f693bfccc1fce90b488f6f62 100644 --- a/data/models/quazim0t0_casa-14b-sce.json +++ b/data/models/quazim0t0_casa-14b-sce.json @@ -5,7 +5,7 @@ "developer": "Quazim0t0", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "LlamaForCausalLM", "params_billions": "14.66" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6718 + "score": 0.6654 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6891 + "score": 0.6901 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4985 + "score": 0.4698 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3339 + "score": 0.3331 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4323 + "score": 0.431 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5408 + "score": 0.5426 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6654 + "score": 0.6718 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6901 + "score": 0.6891 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4698 + "score": 0.4985 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3331 + "score": 0.3339 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.431 + "score": 0.4323 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5426 + "score": 0.5408 } } ], diff --git a/data/models/quazim0t0_odb-14b-sce.json b/data/models/quazim0t0_odb-14b-sce.json index b2854fae2e7ea9ad5a32eb4ec453ef9c1365f1b7..0894dd13c3bc882b56c1b90b77e354b0b8fcf1b3 100644 --- a/data/models/quazim0t0_odb-14b-sce.json +++ b/data/models/quazim0t0_odb-14b-sce.json @@ -6,8 +6,8 @@ "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", - "architecture": "LlamaForCausalLM", - "params_billions": "14.66", + "architecture": "Unknown", + "params_billions": "0.0", "model_id_aliases": [ "Quazim0t0/ODB-14b-sce" ] @@ -15,7 +15,7 @@ }, "evaluations": [ { - "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14b-sce/1773936498.240187", + "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14B-sce/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -47,7 +47,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7016 + "score": 0.2922 } }, { @@ -65,7 +65,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6942 + "score": 0.6559 } }, { @@ -83,7 +83,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4116 + "score": 0.2545 } }, { @@ -101,7 +101,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3624 + "score": 0.2659 } }, { @@ -119,7 +119,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4571 + "score": 0.3929 } }, { @@ -137,7 +137,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5411 + "score": 0.5207 } } ], @@ -145,7 +145,7 @@ "generation_config": null }, { - "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14B-sce/1773936498.240187", + "evaluation_id": "hfopenllm_v2/Quazim0t0_ODB-14b-sce/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", @@ -177,7 +177,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2922 + "score": 0.7016 } }, { @@ -195,7 +195,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6559 + "score": 0.6942 } }, { @@ -213,7 +213,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2545 + "score": 0.4116 } }, { @@ -231,7 +231,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2659 + "score": 0.3624 } }, { @@ -249,7 +249,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3929 + "score": 0.4571 } }, { @@ -267,7 +267,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5207 + "score": 0.5411 } } ], diff --git a/data/models/qwen_qwen2.5-0.5b-instruct.json b/data/models/qwen_qwen2.5-0.5b-instruct.json index 09847e1fd5f51b0203501d1e930a33e69907de86..c0e268fa987678ac81eb89b230a531c20916f2db 100644 --- a/data/models/qwen_qwen2.5-0.5b-instruct.json +++ b/data/models/qwen_qwen2.5-0.5b-instruct.json @@ -5,9 +5,9 @@ "developer": "Qwen", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Qwen2ForCausalLM", - "params_billions": "0.494" + "params_billions": "0.5" } }, "evaluations": [ @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3153 + "score": 0.3071 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3322 + "score": 0.3341 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1035 + "score": 0.0 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2592 + "score": 0.2576 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3342 + "score": 0.3329 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.172 + "score": 0.1697 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3071 + "score": 0.3153 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3341 + "score": 0.3322 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.1035 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2576 + "score": 0.2592 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3329 + "score": 0.3342 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1697 + "score": 0.172 } } ], diff --git a/data/models/ray2333_grm-gemma2-2b-rewardmodel-ft.json b/data/models/ray2333_grm-gemma2-2b-rewardmodel-ft.json index cd8835eaf989dbd622c51262fef9b72b10534d67..24dd55ac376e51965493e81885df295559db7184 100644 --- a/data/models/ray2333_grm-gemma2-2b-rewardmodel-ft.json +++ b/data/models/ray2333_grm-gemma2-2b-rewardmodel-ft.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8839 + "score": 0.5966 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9302 + "score": 0.5305 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7719 + "score": 0.3125 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9216 + "score": 0.5902 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.912 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5966 + "score": 0.7455 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5305 + "score": 0.4788 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Ray2333_GRM-gemma2-2B-rewardmodel-ft/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3125 + "score": 0.8839 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5902 + "score": 0.9302 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.7719 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7455 + "score": 0.9216 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4788 + "score": 0.912 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json b/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json index 294d09da30917c4eff496ef892b03ff9d96db067..569e87e2d1650530a990e95b347a974efc9f47d2 100644 --- a/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json +++ b/data/models/ray2333_grm-llama3-8b-rewardmodel-ft.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9154 + "score": 0.6766 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9553 + "score": 0.6274 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8618 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9081 + "score": 0.5847 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9362 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6766 + "score": 0.8929 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6274 + "score": 0.6824 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Ray2333_GRM-Llama3-8B-rewardmodel-ft/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9154 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5847 + "score": 0.9553 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.8618 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8929 + "score": 0.9081 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6824 + "score": 0.9362 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/ray2333_grm-llama3-8b-sftreg.json b/data/models/ray2333_grm-llama3-8b-sftreg.json index 15f35c842dfd5169bfd85bae53f76a90d5850421..bd70639489991e3e8880cad041284119bd35ecc9 100644 --- a/data/models/ray2333_grm-llama3-8b-sftreg.json +++ b/data/models/ray2333_grm-llama3-8b-sftreg.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6089 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6189 + "score": 0.8542 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3875 + "score": 0.986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5792 + "score": 0.6776 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7867 + "score": 0.8919 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6828 + "score": 0.9229 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5981 + "score": 0.7309 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", + "evaluation_id": "reward-bench-2/Ray2333_GRM-llama3-8B-sftreg/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8542 + "score": 0.6089 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.986 + "score": 0.6189 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6776 + "score": 0.3875 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5792 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8919 + "score": 0.7867 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9229 + "score": 0.6828 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7309 + "score": 0.5981 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json b/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json index 34121094195afd5811c50d769ba5e5e07669b30d..83bfdb05cda442c3bac351d424fa26cf0ae864fe 100644 --- a/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json +++ b/data/models/recoilme_recoilme-gemma-2-9b-v0.2.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2747 + "score": 0.7592 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6031 + "score": 0.6026 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0831 + "score": 0.0529 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3305 + "score": 0.3289 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4686 + "score": 0.4099 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4122 + "score": 0.4163 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7592 + "score": 0.2747 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6026 + "score": 0.6031 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0529 + "score": 0.0831 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3289 + "score": 0.3305 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4099 + "score": 0.4686 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4163 + "score": 0.4122 } } ], diff --git a/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json b/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json index 812871dcfb75c54e3765d2c8d03eeadc60d898aa..d5cadb462419994ffdabe8b6cdbd73be57e99fa8 100644 --- a/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json +++ b/data/models/recoilme_recoilme-gemma-2-9b-v0.3.json @@ -5,7 +5,7 @@ "developer": "recoilme", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "Gemma2ForCausalLM", "params_billions": "10.159" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7439 + "score": 0.5761 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5993 + "score": 0.602 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0876 + "score": 0.1888 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3238 + "score": 0.3372 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4204 + "score": 0.4632 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4072 + "score": 0.4039 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5761 + "score": 0.7439 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.602 + "score": 0.5993 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1888 + "score": 0.0876 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3372 + "score": 0.3238 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4632 + "score": 0.4204 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4039 + "score": 0.4072 } } ], diff --git a/data/models/replete-ai_replete-llm-qwen2-7b.json b/data/models/replete-ai_replete-llm-qwen2-7b.json index e51b10d1bb4ff7d76603042a5cdf25b16b9bc87f..627d67572ee0cd0135e173767115d5ee7360b5e6 100644 --- a/data/models/replete-ai_replete-llm-qwen2-7b.json +++ b/data/models/replete-ai_replete-llm-qwen2-7b.json @@ -5,7 +5,7 @@ "developer": "Replete-AI", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "7.616" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0932 + "score": 0.0905 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2977 + "score": 0.2985 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2475 + "score": 0.2534 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3941 + "score": 0.3848 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1157 + "score": 0.1158 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0905 + "score": 0.0932 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2985 + "score": 0.2977 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.2475 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3848 + "score": 0.3941 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1158 + "score": 0.1157 } } ], diff --git a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json index ecef2ed14b1e2210796bfe9e4943404f7ebd3b7c..83ff30916d3f35c6333d5f652927c45d63836756 100644 --- a/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json +++ b/data/models/sfairxc_fsfairx-llama3-rm-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.6292 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5916 + "score": 0.8338 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4188 + "score": 0.9944 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6284 + "score": 0.6513 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7667 + "score": 0.8676 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7051 + "score": 0.8644 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6647 + "score": 0.7492 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", + "evaluation_id": "reward-bench-2/sfairXC_FsfairX-LLaMA3-RM-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8338 + "score": 0.6292 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9944 + "score": 0.5916 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6513 + "score": 0.4188 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.6284 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8676 + "score": 0.7667 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8644 + "score": 0.7051 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7492 + "score": 0.6647 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json index 988f599afd76311b87e49f16160d0ca534436cdd..4a85b178092e7e5acaf395eb054cb8adc61e2891 100644 --- a/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json +++ b/data/models/shikaichen_ldl-reward-gemma-2-27b-v0.1.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,104 +31,128 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7249 + "score": 0.9499 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7558 + "score": 0.9637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6448 + "score": 0.9378 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9222 + "score": 0.9903 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench-2/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench 2", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Focus", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9131 + "score": 0.7249 }, "source_data": { "dataset_name": "RewardBench 2", @@ -137,135 +161,111 @@ } }, { - "evaluation_name": "Ties", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7633 + "score": 0.7558 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench/ShikaiChen_LDL-Reward-Gemma-2-27B-v0.1/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9499 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9637 + "score": 0.6448 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9079 + "score": 0.9222 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9378 + "score": 0.9131 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9903 + "score": 0.7633 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/skywork_skywork-reward-gemma-2-27b.json b/data/models/skywork_skywork-reward-gemma-2-27b.json index 0b4cc11a7f17a4535cc37f935fdae034a6214bce..c2b742dbc17aa9720f66d090e08eb784ec7accdc 100644 --- a/data/models/skywork_skywork-reward-gemma-2-27b.json +++ b/data/models/skywork_skywork-reward-gemma-2-27b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.938 + "score": 0.7576 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9581 + "score": 0.7368 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9145 + "score": 0.4031 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9189 + "score": 0.7049 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9606 + "score": 0.9422 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7576 + "score": 0.9323 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7368 + "score": 0.8261 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Skywork_Skywork-Reward-Gemma-2-27B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4031 + "score": 0.938 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7049 + "score": 0.9581 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9422 + "score": 0.9145 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9323 + "score": 0.9189 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8261 + "score": 0.9606 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/skywork_skywork-vl-reward-7b.json b/data/models/skywork_skywork-vl-reward-7b.json index d1caca7afd32adac0ce3eaccc5894c6b1d1db99d..651d1416fd84d9618565234fc2f23befa272cb51 100644 --- a/data/models/skywork_skywork-vl-reward-7b.json +++ b/data/models/skywork_skywork-vl-reward-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench/Skywork_Skywork-VL-Reward-7B/1766412838.146816", + "evaluation_id": "reward-bench-2/Skywork_Skywork-VL-Reward-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,128 +31,104 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9007 + "score": 0.6885 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8994 + "score": 0.6063 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.875 + "score": 0.35 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Math", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Math score - measures mathematical reasoning", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9108 + "score": 0.6339 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9176 + "score": 0.8911 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } - } - ], - "detailed_evaluation_results": null, - "generation_config": null - }, - { - "evaluation_id": "reward-bench-2/Skywork_Skywork-VL-Reward-7B/1766412838.146816", - "retrieved_timestamp": "1766412838.146816", - "source_metadata": { - "source_name": "RewardBench 2", - "source_type": "documentation", - "source_organization_name": "Allen Institute for AI", - "source_organization_url": "https://allenai.org", - "evaluator_relationship": "third_party" - }, - "eval_library": { - "name": "rewardbench", - "version": "0.1.3", - "additional_details": { - "subsets": "Chat, Chat Hard, Safety, Reasoning", - "hf_space": "allenai/reward-bench" - } - }, - "benchmark": "reward-bench", - "evaluation_results": [ + }, { - "evaluation_name": "Score", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6885 + "score": 0.8909 }, "source_data": { "dataset_name": "RewardBench 2", @@ -161,111 +137,135 @@ } }, { - "evaluation_name": "Factuality", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6063 + "score": 0.7586 }, "source_data": { "dataset_name": "RewardBench 2", "source_type": "hf_dataset", "hf_repo": "allenai/reward-bench-2-results" } - }, + } + ], + "detailed_evaluation_results": null, + "generation_config": null + }, + { + "evaluation_id": "reward-bench/Skywork_Skywork-VL-Reward-7B/1766412838.146816", + "retrieved_timestamp": "1766412838.146816", + "source_metadata": { + "source_name": "RewardBench", + "source_type": "documentation", + "source_organization_name": "Allen Institute for AI", + "source_organization_url": "https://allenai.org", + "evaluator_relationship": "third_party" + }, + "eval_library": { + "name": "rewardbench", + "version": "0.1.3", + "additional_details": { + "subsets": "Chat, Chat Hard, Safety, Reasoning", + "hf_space": "allenai/reward-bench" + } + }, + "benchmark": "reward-bench", + "evaluation_results": [ { - "evaluation_name": "Precise IF", + "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.35 + "score": 0.9007 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6339 + "score": 0.8994 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Safety", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8911 + "score": 0.875 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8909 + "score": 0.9108 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7586 + "score": 0.9176 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], diff --git a/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json b/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json index 1f18d691fd936d47f7bbe51ac3217aef426c667e..a439bc62eb51d7fa9519bb2716501909fa712711 100644 --- a/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json +++ b/data/models/sometimesanotion_qwenvergence-14b-v3-reason.json @@ -5,7 +5,7 @@ "developer": "sometimesanotion", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Qwen2ForCausalLM", "params_billions": "14.0" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5278 + "score": 0.5367 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6557 + "score": 0.6561 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3119 + "score": 0.358 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3842 + "score": 0.3867 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4754 + "score": 0.474 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5396 + "score": 0.5395 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5367 + "score": 0.5278 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.6561 + "score": 0.6557 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.358 + "score": 0.3119 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3867 + "score": 0.3842 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.474 + "score": 0.4754 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5395 + "score": 0.5396 } } ], diff --git a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json index 7fb384620bce09b34b0d31ff763d2b1e971d0558..5d50cd0f56f0ac33470efc8c81f51c02a702598e 100644 --- a/data/models/tanliboy_lambda-gemma-2-9b-dpo.json +++ b/data/models/tanliboy_lambda-gemma-2-9b-dpo.json @@ -5,7 +5,7 @@ "developer": "tanliboy", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "Gemma2ForCausalLM", "params_billions": "9.242" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1829 + "score": 0.4501 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5488 + "score": 0.5472 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0 + "score": 0.0944 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3104 + "score": 0.3138 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4056 + "score": 0.4017 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3805 + "score": 0.3792 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4501 + "score": 0.1829 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.5472 + "score": 0.5488 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0944 + "score": 0.0 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3138 + "score": 0.3104 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4017 + "score": 0.4056 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3792 + "score": 0.3805 } } ], diff --git a/data/models/valiantlabs_llama3.1-8b-cobalt.json b/data/models/valiantlabs_llama3.1-8b-cobalt.json index c47cbaf67f9d28f7ee6c973b80a6e624aec4a0cc..e1d1a85c95f8deab607165ea40b9f93f0dba9c6d 100644 --- a/data/models/valiantlabs_llama3.1-8b-cobalt.json +++ b/data/models/valiantlabs_llama3.1-8b-cobalt.json @@ -5,7 +5,7 @@ "developer": "ValiantLabs", "inference_platform": "unknown", "additional_details": { - "precision": "float16", + "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.03" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3496 + "score": 0.7168 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4947 + "score": 0.4911 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1269 + "score": 0.1533 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3037 + "score": 0.2861 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3959 + "score": 0.3512 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3644 + "score": 0.3663 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.7168 + "score": 0.3496 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4911 + "score": 0.4947 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1533 + "score": 0.1269 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2861 + "score": 0.3037 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3512 + "score": 0.3959 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3663 + "score": 0.3644 } } ], diff --git a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json index 564c2ceeb0768d947ec7e8507c351558f76e3907..f9d2b14f0a0f79f11e39957c0f38b89aa1e78ac9 100644 --- a/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json +++ b/data/models/weqweasdas_hh_rlhf_rm_open_llama_3b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.2498 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3642 + "score": 0.5027 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.275 + "score": 0.8184 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3497 + "score": 0.3728 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.24 + "score": 0.4149 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2384 + "score": 0.3281 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.0315 + "score": 0.6564 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_hh_rlhf_rm_open_llama_3b/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5027 + "score": 0.2498 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8184 + "score": 0.3642 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3728 + "score": 0.275 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.3497 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.24 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3281 + "score": 0.2384 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6564 + "score": 0.0315 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/weqweasdas_rm-gemma-2b.json b/data/models/weqweasdas_rm-gemma-2b.json index b1151d30882a530616e4e5f2252a198e5a25a8a2..23bdb9776ed416ae7ab56cbae8b9e5655ba27e0c 100644 --- a/data/models/weqweasdas_rm-gemma-2b.json +++ b/data/models/weqweasdas_rm-gemma-2b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-2B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-2B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.3057 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3705 + "score": 0.6549 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2812 + "score": 0.9441 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4317 + "score": 0.4079 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3311 + "score": 0.4986 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.2343 + "score": 0.7637 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.1851 + "score": 0.6652 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/weqweasdas_RM-Gemma-2B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Gemma-2B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6549 + "score": 0.3057 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9441 + "score": 0.3705 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4079 + "score": 0.2812 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.4317 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.4986 + "score": 0.3311 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7637 + "score": 0.2343 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6652 + "score": 0.1851 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/weqweasdas_rm-mistral-7b.json b/data/models/weqweasdas_rm-mistral-7b.json index 2c0c95b4657b4530753b94c6c05b68b220f49072..014b8589e308fa2571a3ca85971364da616341ae 100644 --- a/data/models/weqweasdas_rm-mistral-7b.json +++ b/data/models/weqweasdas_rm-mistral-7b.json @@ -9,10 +9,10 @@ }, "evaluations": [ { - "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench 2", + "source_name": "RewardBench", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -31,127 +31,109 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", - "lower_is_better": false, - "score_type": "continuous", - "min_score": 0.0, - "max_score": 1.0 - }, - "score_details": { - "score": 0.596 - }, - "source_data": { - "dataset_name": "RewardBench 2", - "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" - } - }, - { - "evaluation_name": "Factuality", - "metric_config": { - "evaluation_description": "Factuality score - measures factual accuracy", + "evaluation_description": "Overall RewardBench Score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5937 + "score": 0.7982 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Precise IF", + "evaluation_name": "Chat", "metric_config": { - "evaluation_description": "Precise Instruction Following score", + "evaluation_description": "Chat accuracy - includes easy chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.3438 + "score": 0.9665 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Math", + "evaluation_name": "Chat Hard", "metric_config": { - "evaluation_description": "Math score - measures mathematical reasoning", + "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.5956 + "score": 0.6053 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety score - measures safety awareness", + "evaluation_description": "Safety accuracy - includes safety subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6911 + "score": 0.8703 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Focus", + "evaluation_name": "Reasoning", "metric_config": { - "evaluation_description": "Focus score - measures response focus", + "evaluation_description": "Reasoning accuracy - includes code and math subsets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7293 + "score": 0.7736 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } }, { - "evaluation_name": "Ties", + "evaluation_name": "Prior Sets (0.5 weight)", "metric_config": { - "evaluation_description": "Ties score - ability to identify tie cases", + "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6226 + "score": 0.753 }, "source_data": { - "dataset_name": "RewardBench 2", + "dataset_name": "RewardBench", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench-2-results" + "hf_repo": "allenai/reward-bench" } } ], @@ -159,10 +141,10 @@ "generation_config": null }, { - "evaluation_id": "reward-bench/weqweasdas_RM-Mistral-7B/1766412838.146816", + "evaluation_id": "reward-bench-2/weqweasdas_RM-Mistral-7B/1766412838.146816", "retrieved_timestamp": "1766412838.146816", "source_metadata": { - "source_name": "RewardBench", + "source_name": "RewardBench 2", "source_type": "documentation", "source_organization_name": "Allen Institute for AI", "source_organization_url": "https://allenai.org", @@ -181,109 +163,127 @@ { "evaluation_name": "Score", "metric_config": { - "evaluation_description": "Overall RewardBench Score", + "evaluation_description": "Overall RewardBench 2 Score (mean of all metrics)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7982 + "score": 0.596 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat", + "evaluation_name": "Factuality", "metric_config": { - "evaluation_description": "Chat accuracy - includes easy chat subsets", + "evaluation_description": "Factuality score - measures factual accuracy", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.9665 + "score": 0.5937 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Chat Hard", + "evaluation_name": "Precise IF", "metric_config": { - "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", + "evaluation_description": "Precise Instruction Following score", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.6053 + "score": 0.3438 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" + } + }, + { + "evaluation_name": "Math", + "metric_config": { + "evaluation_description": "Math score - measures mathematical reasoning", + "lower_is_better": false, + "score_type": "continuous", + "min_score": 0.0, + "max_score": 1.0 + }, + "score_details": { + "score": 0.5956 + }, + "source_data": { + "dataset_name": "RewardBench 2", + "source_type": "hf_dataset", + "hf_repo": "allenai/reward-bench-2-results" } }, { "evaluation_name": "Safety", "metric_config": { - "evaluation_description": "Safety accuracy - includes safety subsets", + "evaluation_description": "Safety score - measures safety awareness", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.8703 + "score": 0.6911 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Reasoning", + "evaluation_name": "Focus", "metric_config": { - "evaluation_description": "Reasoning accuracy - includes code and math subsets", + "evaluation_description": "Focus score - measures response focus", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.7736 + "score": 0.7293 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } }, { - "evaluation_name": "Prior Sets (0.5 weight)", + "evaluation_name": "Ties", "metric_config": { - "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", + "evaluation_description": "Ties score - ability to identify tie cases", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { - "score": 0.753 + "score": 0.6226 }, "source_data": { - "dataset_name": "RewardBench", + "dataset_name": "RewardBench 2", "source_type": "hf_dataset", - "hf_repo": "allenai/reward-bench" + "hf_repo": "allenai/reward-bench-2-results" } } ], diff --git a/data/models/yam-peleg_hebrew-mistral-7b-200k.json b/data/models/yam-peleg_hebrew-mistral-7b-200k.json index 1a70a6eb9b375af7d3ae622e407043222ccb78a0..baae674a05af479a549c721b22d1c0240e173193 100644 --- a/data/models/yam-peleg_hebrew-mistral-7b-200k.json +++ b/data/models/yam-peleg_hebrew-mistral-7b-200k.json @@ -5,7 +5,7 @@ "developer": "yam-peleg", "inference_platform": "unknown", "additional_details": { - "precision": "bfloat16", + "precision": "float16", "architecture": "MistralForCausalLM", "params_billions": "7.504" } @@ -44,7 +44,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.177 + "score": 0.1856 } }, { @@ -62,7 +62,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3411 + "score": 0.4149 } }, { @@ -80,7 +80,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.031 + "score": 0.0234 } }, { @@ -98,7 +98,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2534 + "score": 0.276 } }, { @@ -116,7 +116,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.374 + "score": 0.3765 } }, { @@ -134,7 +134,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2529 + "score": 0.2573 } } ], @@ -174,7 +174,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.1856 + "score": 0.177 } }, { @@ -192,7 +192,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.4149 + "score": 0.3411 } }, { @@ -210,7 +210,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.0234 + "score": 0.031 } }, { @@ -228,7 +228,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.276 + "score": 0.2534 } }, { @@ -246,7 +246,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.3765 + "score": 0.374 } }, { @@ -264,7 +264,7 @@ "max_score": 1.0 }, "score_details": { - "score": 0.2573 + "score": 0.2529 } } ], diff --git a/data/models/zhipu-ai_glm-4.7.json b/data/models/zhipu-ai_glm-4.7.json index ef6efd4fd9091259b58749fefdf86e2517723096..9ffeed9fa5797e0518c941376cc795f418b5ed1c 100644 --- a/data/models/zhipu-ai_glm-4.7.json +++ b/data/models/zhipu-ai_glm-4.7.json @@ -4,13 +4,13 @@ "id": "zhipu-ai/glm-4.7", "developer": "Z-AI", "additional_details": { - "agent_name": "Terminus 2", - "agent_organization": "Terminal Bench" + "agent_name": "Crux", + "agent_organization": "Roam" } }, "evaluations": [ { - "evaluation_id": "terminal-bench-2.0/terminus-2__glm-4.7/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/crux__glm-4.7/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -34,7 +34,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-01-28", + "evaluation_timestamp": "2026-02-08", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -43,17 +43,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.4, + "score": 33.3, "uncertainty": { "standard_error": { - "value": 2.8 + "value": 2.5 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -70,7 +70,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -84,7 +84,7 @@ } }, { - "evaluation_id": "terminal-bench-2.0/crux__glm-4.7/1773776901.772108", + "evaluation_id": "terminal-bench-2.0/terminus-2__glm-4.7/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", @@ -108,7 +108,7 @@ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, - "evaluation_timestamp": "2026-02-08", + "evaluation_timestamp": "2026-01-28", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, @@ -117,17 +117,17 @@ "max_score": 100.0 }, "score_details": { - "score": 33.3, + "score": 33.4, "uncertainty": { "standard_error": { - "value": 2.5 + "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ { @@ -144,7 +144,7 @@ "detailed_evaluation_results": null, "generation_config": { "generation_args": { - "execution_command": "harbor run -d terminal-bench@2.0 -a \"Crux\" -m \"GLM 4.7\" -k 5", + "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"GLM 4.7\" -k 5", "agentic_eval_config": { "available_tools": [ {