Spaces:
Running
Running
| { | |
| "model_info": { | |
| "name": "0-hero/Matter-0.1-7B-boost-DPO-preview", | |
| "id": "0-hero/Matter-0.1-7B-boost-DPO-preview", | |
| "developer": "0-hero", | |
| "additional_details": { | |
| "model_type": "DPO" | |
| } | |
| }, | |
| "evaluations": [ | |
| { | |
| "evaluation_id": "reward-bench/0-hero_Matter-0.1-7B-boost-DPO-preview/1766412838.146816", | |
| "retrieved_timestamp": "1766412838.146816", | |
| "source_metadata": { | |
| "source_name": "RewardBench", | |
| "source_type": "documentation", | |
| "source_organization_name": "Allen Institute for AI", | |
| "source_organization_url": "https://allenai.org", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "eval_library": { | |
| "name": "rewardbench", | |
| "version": "0.1.3", | |
| "additional_details": { | |
| "subsets": "Chat, Chat Hard, Safety, Reasoning", | |
| "hf_space": "allenai/reward-bench" | |
| } | |
| }, | |
| "benchmark": "reward-bench", | |
| "evaluation_results": [ | |
| { | |
| "evaluation_name": "Score", | |
| "metric_config": { | |
| "evaluation_description": "Overall RewardBench Score", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.7448 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| }, | |
| { | |
| "evaluation_name": "Chat", | |
| "metric_config": { | |
| "evaluation_description": "Chat accuracy - includes easy chat subsets", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.9106 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| }, | |
| { | |
| "evaluation_name": "Chat Hard", | |
| "metric_config": { | |
| "evaluation_description": "Chat Hard accuracy - includes hard chat subsets", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.6096 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| }, | |
| { | |
| "evaluation_name": "Safety", | |
| "metric_config": { | |
| "evaluation_description": "Safety accuracy - includes safety subsets", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.7135 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| }, | |
| { | |
| "evaluation_name": "Reasoning", | |
| "metric_config": { | |
| "evaluation_description": "Reasoning accuracy - includes code and math subsets", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.8395 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| }, | |
| { | |
| "evaluation_name": "Prior Sets (0.5 weight)", | |
| "metric_config": { | |
| "evaluation_description": "Prior Sets score (weighted 0.5) - includes test sets", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.5566 | |
| }, | |
| "source_data": { | |
| "dataset_name": "RewardBench", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "allenai/reward-bench" | |
| } | |
| } | |
| ], | |
| "detailed_evaluation_results": null, | |
| "generation_config": null | |
| } | |
| ] | |
| } |