general-eval-card / data /models /152334h_miqu-1-70b-sf.json
GitHub Actions
chore: sync EEE pipeline output [2026-03-28 11:49 UTC]
d91b463
Raw
History Blame
4.26 kB
{
"model_info": {
"name": "miqu-1-70b-sf",
"id": "152334H/miqu-1-70b-sf",
"developer": "152334H",
"inference_platform": "unknown",
"additional_details": {
"precision": "float16",
"architecture": "LlamaForCausalLM",
"params_billions": "68.977"
}
},
"evaluations": [
{
"evaluation_id": "hfopenllm_v2/152334H_miqu-1-70b-sf/1773936498.240187",
"retrieved_timestamp": "1773936498.240187",
"source_metadata": {
"source_name": "HF Open LLM v2",
"source_type": "documentation",
"source_organization_name": "Hugging Face",
"evaluator_relationship": "third_party"
},
"eval_library": {
"name": "lm-evaluation-harness",
"version": "0.4.0",
"additional_details": {
"fork": "https://github.com/huggingface/lm-evaluation-harness/tree/adding_all_changess"
}
},
"benchmark": "hfopenllm_v2",
"evaluation_results": [
{
"evaluation_name": "IFEval",
"source_data": {
"dataset_name": "IFEval",
"source_type": "hf_dataset",
"hf_repo": "google/IFEval"
},
"metric_config": {
"evaluation_description": "Accuracy on IFEval",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.5182
}
},
{
"evaluation_name": "BBH",
"source_data": {
"dataset_name": "BBH",
"source_type": "hf_dataset",
"hf_repo": "SaylorTwift/bbh"
},
"metric_config": {
"evaluation_description": "Accuracy on BBH",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.6102
}
},
{
"evaluation_name": "MATH Level 5",
"source_data": {
"dataset_name": "MATH Level 5",
"source_type": "hf_dataset",
"hf_repo": "DigitalLearningGmbH/MATH-lighteval"
},
"metric_config": {
"evaluation_description": "Exact Match on MATH Level 5",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.1246
}
},
{
"evaluation_name": "GPQA",
"source_data": {
"dataset_name": "GPQA",
"source_type": "hf_dataset",
"hf_repo": "Idavidrein/gpqa"
},
"metric_config": {
"evaluation_description": "Accuracy on GPQA",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.3507
}
},
{
"evaluation_name": "MUSR",
"source_data": {
"dataset_name": "MUSR",
"source_type": "hf_dataset",
"hf_repo": "TAUR-Lab/MuSR"
},
"metric_config": {
"evaluation_description": "Accuracy on MUSR",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.4582
}
},
{
"evaluation_name": "MMLU-PRO",
"source_data": {
"dataset_name": "MMLU-PRO",
"source_type": "hf_dataset",
"hf_repo": "TIGER-Lab/MMLU-Pro"
},
"metric_config": {
"evaluation_description": "Accuracy on MMLU-PRO",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"score_details": {
"score": 0.4228
}
}
],
"detailed_evaluation_results": null,
"generation_config": null
}
]
}