general-eval-card / data /models /0-hero_matter-0.1-7b-boost-dpo-preview.json
GitHub Actions
chore: sync EEE pipeline output [2026-03-28 11:49 UTC]
d91b463
Raw
History Blame
4.47 kB
{
"model_info": {
"name": "0-hero/Matter-0.1-7B-boost-DPO-preview",
"id": "0-hero/Matter-0.1-7B-boost-DPO-preview",
"developer": "0-hero",
"additional_details": {
"model_type": "DPO"
}
},
"evaluations": [
{
"evaluation_id": "reward-bench/0-hero_Matter-0.1-7B-boost-DPO-preview/1766412838.146816",
"retrieved_timestamp": "1766412838.146816",
"source_metadata": {
"source_name": "RewardBench",
"source_type": "documentation",
"source_organization_name": "Allen Institute for AI",
"source_organization_url": "https://allenai.org",
"evaluator_relationship": "third_party"
},
"eval_library": {
"name": "rewardbench",
"version": "0.1.3",
"additional_details": {
"subsets": "Chat, Chat Hard, Safety, Reasoning",
"hf_space": "allenai/reward-bench"
}
},
"benchmark": "reward-bench",
"evaluation_results": [
{
"evaluation_name": "Score",
"metric_config": {
"evaluation_description": "Overall RewardBench Score",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.7448
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
},
{
"evaluation_name": "Chat",
"metric_config": {
"evaluation_description": "Chat accuracy - includes easy chat subsets",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.9106
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
},
{
"evaluation_name": "Chat Hard",
"metric_config": {
"evaluation_description": "Chat Hard accuracy - includes hard chat subsets",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.6096
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
},
{
"evaluation_name": "Safety",
"metric_config": {
"evaluation_description": "Safety accuracy - includes safety subsets",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.7135
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
},
{
"evaluation_name": "Reasoning",
"metric_config": {
"evaluation_description": "Reasoning accuracy - includes code and math subsets",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.8395
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
},
{
"evaluation_name": "Prior Sets (0.5 weight)",
"metric_config": {
"evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.5566
},
"source_data": {
"dataset_name": "RewardBench",
"source_type": "hf_dataset",
"hf_repo": "allenai/reward-bench"
}
}
],
"detailed_evaluation_results": null,
"generation_config": null
}
]
}