Spaces:
Running
Running
| { | |
| "model_info": { | |
| "name": "T-VisStar-7B-v0.1", | |
| "id": "1TuanPham/T-VisStar-7B-v0.1", | |
| "developer": "1TuanPham", | |
| "inference_platform": "unknown", | |
| "additional_details": { | |
| "precision": "float16", | |
| "architecture": "MistralForCausalLM", | |
| "params_billions": "7.294" | |
| } | |
| }, | |
| "evaluations": [ | |
| { | |
| "evaluation_id": "hfopenllm_v2/1TuanPham_T-VisStar-7B-v0.1/1773936498.240187", | |
| "retrieved_timestamp": "1773936498.240187", | |
| "source_metadata": { | |
| "source_name": "HF Open LLM v2", | |
| "source_type": "documentation", | |
| "source_organization_name": "Hugging Face", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "eval_library": { | |
| "name": "lm-evaluation-harness", | |
| "version": "0.4.0", | |
| "additional_details": { | |
| "fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess" | |
| } | |
| }, | |
| "benchmark": "hfopenllm_v2", | |
| "evaluation_results": [ | |
| { | |
| "evaluation_name": "IFEval", | |
| "source_data": { | |
| "dataset_name": "IFEval", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "google/IFEval" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Accuracy on IFEval", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.3607 | |
| } | |
| }, | |
| { | |
| "evaluation_name": "BBH", | |
| "source_data": { | |
| "dataset_name": "BBH", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "SaylorTwift/bbh" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Accuracy on BBH", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.5052 | |
| } | |
| }, | |
| { | |
| "evaluation_name": "MATH Level 5", | |
| "source_data": { | |
| "dataset_name": "MATH Level 5", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "DigitalLearningGmbH/MATH-lighteval" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Exact Match on MATH Level 5", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.0574 | |
| } | |
| }, | |
| { | |
| "evaluation_name": "GPQA", | |
| "source_data": { | |
| "dataset_name": "GPQA", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "Idavidrein/gpqa" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Accuracy on GPQA", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.2852 | |
| } | |
| }, | |
| { | |
| "evaluation_name": "MUSR", | |
| "source_data": { | |
| "dataset_name": "MUSR", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "TAUR-Lab/MuSR" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Accuracy on MUSR", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.4375 | |
| } | |
| }, | |
| { | |
| "evaluation_name": "MMLU-PRO", | |
| "source_data": { | |
| "dataset_name": "MMLU-PRO", | |
| "source_type": "hf_dataset", | |
| "hf_repo": "TIGER-Lab/MMLU-Pro" | |
| }, | |
| "metric_config": { | |
| "evaluation_description": "Accuracy on MMLU-PRO", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "score_details": { | |
| "score": 0.3211 | |
| } | |
| } | |
| ], | |
| "detailed_evaluation_results": null, | |
| "generation_config": null | |
| } | |
| ] | |
| } |