{ "model_info": { "name": "Yi-1.5-9B-32K", "id": "01-ai/Yi-1.5-9B-32K", "developer": "01-ai", "inference_platform": "unknown", "additional_details": { "precision": "bfloat16", "architecture": "LlamaForCausalLM", "params_billions": "8.829" } }, "evaluations": [ { "evaluation_id": "hfopenllm_v2/01-ai_Yi-1.5-9B-32K/1773936498.240187", "retrieved_timestamp": "1773936498.240187", "source_metadata": { "source_name": "HF Open LLM v2", "source_type": "documentation", "source_organization_name": "Hugging Face", "evaluator_relationship": "third_party" }, "eval_library": { "name": "lm-evaluation-harness", "version": "0.4.0", "additional_details": { "fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess" } }, "benchmark": "hfopenllm_v2", "evaluation_results": [ { "evaluation_name": "IFEval", "source_data": { "dataset_name": "IFEval", "source_type": "hf_dataset", "hf_repo": "google/IFEval" }, "metric_config": { "evaluation_description": "Accuracy on IFEval", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.2303 } }, { "evaluation_name": "BBH", "source_data": { "dataset_name": "BBH", "source_type": "hf_dataset", "hf_repo": "SaylorTwift/bbh" }, "metric_config": { "evaluation_description": "Accuracy on BBH", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.4963 } }, { "evaluation_name": "MATH Level 5", "source_data": { "dataset_name": "MATH Level 5", "source_type": "hf_dataset", "hf_repo": "DigitalLearningGmbH/MATH-lighteval" }, "metric_config": { "evaluation_description": "Exact Match on MATH Level 5", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.108 } }, { "evaluation_name": "GPQA", "source_data": { "dataset_name": "GPQA", "source_type": "hf_dataset", "hf_repo": "Idavidrein/gpqa" }, "metric_config": { "evaluation_description": "Accuracy on GPQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.3591 } }, { "evaluation_name": "MUSR", "source_data": { "dataset_name": "MUSR", "source_type": "hf_dataset", "hf_repo": "TAUR-Lab/MuSR" }, "metric_config": { "evaluation_description": "Accuracy on MUSR", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.4186 } }, { "evaluation_name": "MMLU-PRO", "source_data": { "dataset_name": "MMLU-PRO", "source_type": "hf_dataset", "hf_repo": "TIGER-Lab/MMLU-Pro" }, "metric_config": { "evaluation_description": "Accuracy on MMLU-PRO", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "score_details": { "score": 0.3765 } } ], "detailed_evaluation_results": null, "generation_config": null } ] }