{ "model_info": { "name": "GPT-5.2 Pro", "id": "openai/gpt-5-2-pro", "developer": "openai", "additional_details": { "raw_id": "gpt-5.2-pro-2025-12-11", "raw_name": "GPT-5.2 Pro", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_model_name": "GPT-5.2 Pro", "raw_organization_id": "openai", "raw_organization_name": "OpenAI", "raw_context_window": "400000", "raw_input_cost_per_million": "21.0", "raw_output_cost_per_million": "168.0", "raw_release_date": "2025-12-11", "raw_announcement_date": "2025-12-11", "raw_multimodal": "true", "raw_provider_slug": "openai", "raw_provider_name": "OpenAI" }, "normalized_id": "openai/gpt-5.2-pro-2025-12-11", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "GPT-5.2 Pro", "variant_key": "default", "variant_label": "Default", "model_route_id": "openai__gpt-5-2-pro", "model_version": null }, "model_group_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_family_name": "GPT-5.2 Pro", "raw_model_ids": [ "openai/GPT 5.2 Pro", "openai/gpt-5-2-pro-2025-12-11-high", "openai/gpt-5-2-pro-2025-12-11-medium", "openai/gpt-5-2-pro-2025-12-11-xhigh", "openai/gpt-5.2-pro-2025-12-11" ], "evaluations_by_category": { "other": [ { "schema_version": "0.2.2", "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "benchmark": "apex-v1", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "eval_library": { "name": "archipelago", "version": "1.0.0" }, "model_info": { "name": "GPT 5.2 Pro", "developer": "openai", "id": "openai/GPT 5.2 Pro", "inference_platform": "unknown", "normalized_id": "openai/GPT 5.2 Pro", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "GPT 5.2 Pro", "variant_key": "default", "variant_label": "Default", "model_route_id": "openai__gpt-5-2-pro" }, "generation_config": { "additional_details": { "run_setting": "High" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_name": "apex-v1", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Overall APEX-v1 mean score across all jobs.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.668, "uncertainty": { "confidence_interval": { "lower": -0.026, "upper": 0.026, "method": "bootstrap" } } }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/openai_gpt-5.2-pro/1773260200#apex_v1#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false } }, { "evaluation_name": "Consulting", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.64 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/openai_gpt-5.2-pro/1773260200#consulting#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false } }, { "evaluation_name": "Medicine (MD)", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Primary care physician (MD) score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Medicine (MD) Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.65 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/openai_gpt-5.2-pro/1773260200#medicine_md#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "medicine_md", "slice_name": "Medicine (MD)", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Medicine (MD) / Score", "canonical_display_name": "APEX v1 / Medicine (MD) / Score", "raw_evaluation_name": "Medicine (MD)", "is_summary_score": false } }, { "evaluation_name": "Investment Banking", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Investment banking associate score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.64 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/openai_gpt-5.2-pro/1773260200#investment_banking#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Investment Banking / Score", "canonical_display_name": "APEX v1 / Investment Banking / Score", "raw_evaluation_name": "Investment Banking", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "apex_v1" ] }, { "schema_version": "0.2.2", "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "benchmark": "arc-agi", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "eval_library": { "name": "ARC Prize leaderboard", "version": "unknown" }, "model_info": { "name": "gpt-5-2-pro-2025-12-11-high", "id": "openai/gpt-5-2-pro-2025-12-11-high", "developer": "openai", "additional_details": { "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" }, "normalized_id": "openai/gpt-5-2-pro-2025-12-11-high", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "gpt-5-2-pro-2025-12-11-high", "variant_key": "2025-12-11-high", "variant_label": "2025-12-11 high", "model_route_id": "openai__gpt-5-2-pro" }, "generation_config": null, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v1_public_eval#score", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.9462, "details": { "datasetId": "v1_Public_Eval", "costPerTask": "4.6384", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v1_public_eval#cost_per_task", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 4.6384, "details": { "datasetId": "v1_Public_Eval", "score": "0.9462", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v1_semi_private#score", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.8567, "details": { "datasetId": "v1_Semi_Private", "costPerTask": "5.8694", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v1_semi_private#cost_per_task", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 5.8694, "details": { "datasetId": "v1_Semi_Private", "score": "0.8567", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v2_public_eval#score", "evaluation_name": "v2_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.5168, "details": { "datasetId": "v2_Public_Eval", "costPerTask": "16.662", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v2_public_eval#cost_per_task", "evaluation_name": "v2_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 16.662, "details": { "datasetId": "v2_Public_Eval", "score": "0.5168", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v2_semi_private#score", "evaluation_name": "v2_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.5416, "details": { "datasetId": "v2_Semi_Private", "costPerTask": "15.721", "resultsUrl": "", "display": "True", "labelOffsetX": "-2", "labelOffsetY": "-10", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104#v2_semi_private#cost_per_task", "evaluation_name": "v2_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 15.721, "details": { "datasetId": "v2_Semi_Private", "score": "0.5416", "resultsUrl": "", "display": "True", "labelOffsetX": "-2", "labelOffsetY": "-10", "raw_model_id": "gpt-5-2-pro-2025-12-11-high", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-high\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "arc_agi" ] }, { "schema_version": "0.2.2", "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "benchmark": "arc-agi", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "eval_library": { "name": "ARC Prize leaderboard", "version": "unknown" }, "model_info": { "name": "gpt-5-2-pro-2025-12-11-medium", "id": "openai/gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "additional_details": { "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" }, "normalized_id": "openai/gpt-5-2-pro-2025-12-11-medium", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "gpt-5-2-pro-2025-12-11-medium", "variant_key": "2025-12-11-medium", "variant_label": "2025-12-11 medium", "model_route_id": "openai__gpt-5-2-pro" }, "generation_config": null, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v1_public_eval#score", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.9025, "details": { "datasetId": "v1_Public_Eval", "costPerTask": "3.2418", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v1_public_eval#cost_per_task", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 3.2418, "details": { "datasetId": "v1_Public_Eval", "score": "0.9025", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v1_semi_private#score", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.8117, "details": { "datasetId": "v1_Semi_Private", "costPerTask": "3.9774", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v1_semi_private#cost_per_task", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 3.9774, "details": { "datasetId": "v1_Semi_Private", "score": "0.8117", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v2_public_eval#score", "evaluation_name": "v2_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.3792, "details": { "datasetId": "v2_Public_Eval", "costPerTask": "9.5162", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v2_public_eval#cost_per_task", "evaluation_name": "v2_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 9.5162, "details": { "datasetId": "v2_Public_Eval", "score": "0.3792", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v2_semi_private#score", "evaluation_name": "v2_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.3847, "details": { "datasetId": "v2_Semi_Private", "costPerTask": "8.9928", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366#v2_semi_private#cost_per_task", "evaluation_name": "v2_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 8.9928, "details": { "datasetId": "v2_Semi_Private", "score": "0.3847", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-medium", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-medium\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "arc_agi" ] }, { "schema_version": "0.2.2", "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "benchmark": "arc-agi", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "eval_library": { "name": "ARC Prize leaderboard", "version": "unknown" }, "model_info": { "name": "gpt-5-2-pro-2025-12-11-xhigh", "id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "additional_details": { "raw_model_id": "gpt-5-2-pro-2025-12-11-xhigh", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-xhigh\"]" }, "normalized_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "gpt-5-2-pro-2025-12-11-xhigh", "variant_key": "2025-12-11-xhigh", "variant_label": "2025-12-11 xhigh", "model_route_id": "openai__gpt-5-2-pro" }, "generation_config": null, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698#v1_public_eval#score", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.9761, "details": { "datasetId": "v1_Public_Eval", "costPerTask": "7.7201", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-xhigh", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-xhigh\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698#v1_public_eval#cost_per_task", "evaluation_name": "v1_Public_Eval", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 7.7201, "details": { "datasetId": "v1_Public_Eval", "score": "0.9761", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-xhigh", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-xhigh\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698#v1_semi_private#score", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "score_details": { "score": 0.905, "details": { "datasetId": "v1_Semi_Private", "costPerTask": "11.6542", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-xhigh", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-xhigh\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698#v1_semi_private#cost_per_task", "evaluation_name": "v1_Semi_Private", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "score_details": { "score": 11.6542, "details": { "datasetId": "v1_Semi_Private", "score": "0.905", "resultsUrl": "", "display": "True", "raw_model_id": "gpt-5-2-pro-2025-12-11-xhigh", "raw_model_aliases_json": "[\"gpt-5-2-pro-2025-12-11-xhigh\"]" } }, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "arc_agi" ] }, { "schema_version": "0.2.2", "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "benchmark": "llm-stats", "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "eval_library": { "name": "LLM Stats", "version": "unknown" }, "model_info": { "name": "GPT-5.2 Pro", "id": "openai/gpt-5.2-pro-2025-12-11", "developer": "openai", "additional_details": { "raw_id": "gpt-5.2-pro-2025-12-11", "raw_name": "GPT-5.2 Pro", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_model_name": "GPT-5.2 Pro", "raw_organization_id": "openai", "raw_organization_name": "OpenAI", "raw_context_window": "400000", "raw_input_cost_per_million": "21.0", "raw_output_cost_per_million": "168.0", "raw_release_date": "2025-12-11", "raw_announcement_date": "2025-12-11", "raw_multimodal": "true", "raw_provider_slug": "openai", "raw_provider_name": "OpenAI" }, "normalized_id": "openai/gpt-5.2-pro-2025-12-11", "family_id": "openai/gpt-5-2-pro", "family_slug": "gpt-5-2-pro", "family_name": "GPT-5.2 Pro", "variant_key": "2025-12-11", "variant_label": "2025-12-11", "model_route_id": "openai__gpt-5-2-pro" }, "generation_config": null, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_result_id": "aime-2025::aime-2025-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.aime-2025", "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "metric_id": "llm_stats.aime-2025.score", "metric_name": "AIME 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "aime-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "AIME 2025", "raw_categories": "[\"math\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "107" } }, "score_details": { "score": 1.0, "details": { "raw_score": "1.0", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "aime-2025", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/aime-2025\",\"https://api.llm-stats.com/leaderboard/benchmarks/aime-2025\"]", "raw_score_id": "aime-2025::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "AIME 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "AIME 2025", "benchmark_component_key": "aime_2025", "benchmark_component_name": "Aime 2025", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.aime-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Aime 2025 / Score", "canonical_display_name": "Aime 2025 / Score", "raw_evaluation_name": "llm_stats.aime-2025", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi-v2::arc-agi-v2-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.arc-agi-v2", "source_data": { "dataset_name": "ARC-AGI v2", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/arc-agi-v2", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi-v2" ], "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.", "metric_id": "llm_stats.arc-agi-v2.score", "metric_name": "ARC-AGI v2 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI v2", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "multimodal", "raw_verified": "false", "raw_model_count": "15" } }, "score_details": { "score": 0.542, "details": { "raw_score": "0.542", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "arc-agi-v2", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/arc-agi-v2\",\"https://api.llm-stats.com/leaderboard/benchmarks/arc-agi-v2\"]", "raw_score_id": "arc-agi-v2::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI v2", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI v2", "benchmark_component_key": "arc_agi_v2", "benchmark_component_name": "Arc Agi V2", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi-v2.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi V2 / Score", "canonical_display_name": "Arc Agi V2 / Score", "raw_evaluation_name": "llm_stats.arc-agi-v2", "is_summary_score": false } }, { "evaluation_result_id": "arc-agi::arc-agi-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.arc-agi", "source_data": { "dataset_name": "ARC-AGI", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/arc-agi", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi" ], "additional_details": { "raw_benchmark_id": "arc-agi", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.", "metric_id": "llm_stats.arc-agi.score", "metric_name": "ARC-AGI score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "image", "raw_verified": "false", "raw_model_count": "7" } }, "score_details": { "score": 0.905, "details": { "raw_score": "0.905", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "arc-agi", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/arc-agi\",\"https://api.llm-stats.com/leaderboard/benchmarks/arc-agi\"]", "raw_score_id": "arc-agi::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI", "benchmark_component_key": "arc_agi", "benchmark_component_name": "Arc Agi", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi / Score", "canonical_display_name": "Arc Agi / Score", "raw_evaluation_name": "llm_stats.arc-agi", "is_summary_score": false } }, { "evaluation_result_id": "browsecomp::browsecomp-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.browsecomp", "source_data": { "dataset_name": "BrowseComp", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/browsecomp", "https://api.llm-stats.com/leaderboard/benchmarks/browsecomp" ], "additional_details": { "raw_benchmark_id": "browsecomp", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "BrowseComp is a benchmark comprising 1,266 questions that challenge AI agents to persistently navigate the internet in search of hard-to-find, entangled information. The benchmark measures agents' ability to exercise persistence in information gathering, demonstrate creativity in web navigation, and find concise, verifiable answers. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers.", "metric_id": "llm_stats.browsecomp.score", "metric_name": "BrowseComp score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "browsecomp", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "BrowseComp", "raw_categories": "[\"agents\",\"reasoning\",\"search\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "45" } }, "score_details": { "score": 0.779, "details": { "raw_score": "0.779", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "browsecomp", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/browsecomp\",\"https://api.llm-stats.com/leaderboard/benchmarks/browsecomp\"]", "raw_score_id": "browsecomp::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "BrowseComp", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "BrowseComp", "benchmark_component_key": "browsecomp", "benchmark_component_name": "Browsecomp", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.browsecomp.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Browsecomp / Score", "canonical_display_name": "Browsecomp / Score", "raw_evaluation_name": "llm_stats.browsecomp", "is_summary_score": false } }, { "evaluation_result_id": "gpqa::gpqa-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.gpqa", "source_data": { "dataset_name": "GPQA", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/gpqa", "https://api.llm-stats.com/leaderboard/benchmarks/gpqa" ], "additional_details": { "raw_benchmark_id": "gpqa", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Questions are Google-proof and extremely difficult, with PhD experts reaching 65% accuracy.", "metric_id": "llm_stats.gpqa.score", "metric_name": "GPQA score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "gpqa", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "GPQA", "raw_categories": "[\"biology\",\"chemistry\",\"general\",\"physics\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "213" } }, "score_details": { "score": 0.932, "details": { "raw_score": "0.932", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "gpqa", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/gpqa\",\"https://api.llm-stats.com/leaderboard/benchmarks/gpqa\"]", "raw_score_id": "gpqa::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "GPQA", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "GPQA", "benchmark_component_key": "gpqa", "benchmark_component_name": "Gpqa", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.gpqa.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gpqa / Score", "canonical_display_name": "Gpqa / Score", "raw_evaluation_name": "llm_stats.gpqa", "is_summary_score": false } }, { "evaluation_result_id": "hmmt-2025::hmmt-2025-gpt-5.2-pro-2025-12-11", "evaluation_name": "llm_stats.hmmt-2025", "source_data": { "dataset_name": "HMMT 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/hmmt-2025", "https://api.llm-stats.com/leaderboard/benchmarks/hmmt-2025" ], "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "metric_config": { "evaluation_description": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "metric_id": "llm_stats.hmmt-2025.score", "metric_name": "HMMT 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "HMMT 2025", "raw_categories": "[\"math\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "32" } }, "score_details": { "score": 1.0, "details": { "raw_score": "1.0", "raw_score_field": "score", "raw_model_id": "gpt-5.2-pro-2025-12-11", "raw_benchmark_id": "hmmt-2025", "source_urls_json": "[\"https://llm-stats.com/models/gpt-5.2-pro-2025-12-11\",\"https://llm-stats.com/benchmarks/hmmt-2025\",\"https://api.llm-stats.com/leaderboard/benchmarks/hmmt-2025\"]", "raw_score_id": "hmmt-2025::gpt-5.2-pro-2025-12-11", "raw_provenance_label": "unknown", "raw_verified": "false" } }, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "HMMT 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "HMMT 2025", "benchmark_component_key": "hmmt_2025", "benchmark_component_name": "Hmmt 2025", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.hmmt-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Hmmt 2025 / Score", "canonical_display_name": "Hmmt 2025 / Score", "raw_evaluation_name": "llm_stats.hmmt-2025", "is_summary_score": false } } ], "benchmark_card": null, "instance_level_data": null, "eval_summary_ids": [ "llm_stats_aime_2025", "llm_stats_arc_agi", "llm_stats_arc_agi_v2", "llm_stats_browsecomp", "llm_stats_gpqa", "llm_stats_hmmt_2025" ] } ] }, "evaluation_summaries_by_category": { "knowledge": [ { "eval_summary_id": "apex_v1", "benchmark": "APEX v1", "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "evaluation_name": "APEX v1", "display_name": "APEX v1", "canonical_display_name": "APEX v1", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ] }, "subtasks_count": 3, "metrics_count": 4, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 4, "has_reproducibility_gap_count": 4, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 4, "total_groups": 4, "multi_source_groups": 0, "first_party_only_groups": 4, "source_type_distribution": { "first_party": 4, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 4, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_v1_score", "legacy_eval_summary_id": "apex_v1_apex_v1", "evaluation_name": "apex-v1", "display_name": "APEX v1 / Score", "canonical_display_name": "APEX v1 / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall APEX-v1 mean score (paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.668, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.668, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "consulting", "subtask_name": "Consulting", "display_name": "Consulting", "metrics": [ { "metric_summary_id": "apex_v1_consulting_score", "legacy_eval_summary_id": "apex_v1_consulting", "evaluation_name": "Consulting", "display_name": "APEX v1 / Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.64, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] }, { "subtask_key": "investment_banking", "subtask_name": "Investment Banking", "display_name": "Investment Banking", "metrics": [ { "metric_summary_id": "apex_v1_investment_banking_score", "legacy_eval_summary_id": "apex_v1_investment_banking", "evaluation_name": "Investment Banking", "display_name": "APEX v1 / Investment Banking / Score", "canonical_display_name": "APEX v1 / Investment Banking / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "investment_banking", "slice_name": "Investment Banking", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Investment banking associate score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.64, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Investment Banking / Score", "canonical_display_name": "APEX v1 / Investment Banking / Score", "raw_evaluation_name": "Investment Banking", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] }, { "subtask_key": "medicine_md", "subtask_name": "Medicine (MD)", "display_name": "Medicine (MD)", "metrics": [ { "metric_summary_id": "apex_v1_medicine_md_score", "legacy_eval_summary_id": "apex_v1_medicine_md", "evaluation_name": "Medicine (MD)", "display_name": "APEX v1 / Medicine (MD) / Score", "canonical_display_name": "APEX v1 / Medicine (MD) / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "medicine_md", "slice_name": "Medicine (MD)", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Primary care physician (MD) score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Medicine (MD) Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.65, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.65, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "medicine_md", "slice_name": "Medicine (MD)", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Medicine (MD) / Score", "canonical_display_name": "APEX v1 / Medicine (MD) / Score", "raw_evaluation_name": "Medicine (MD)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "other": [ { "eval_summary_id": "llm_stats_aime_2025", "benchmark": "AIME 2025", "benchmark_family_key": "llm_stats", "benchmark_family_name": "AIME 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "AIME 2025", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "benchmark_component_key": "aime_2025", "benchmark_component_name": "Aime 2025", "evaluation_name": "Aime 2025", "display_name": "Aime 2025", "canonical_display_name": "Aime 2025", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-opus-4-6", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "claude-opus-4-6", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_aime_2025_score", "legacy_eval_summary_id": "llm_stats_llm_stats_aime_2025", "evaluation_name": "llm_stats.aime-2025", "display_name": "Aime 2025 / Score", "canonical_display_name": "Aime 2025 / Score", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.aime-2025.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "metric_id": "llm_stats.aime-2025.score", "metric_name": "AIME 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "aime-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "AIME 2025", "raw_categories": "[\"math\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "107" } }, "models_count": 1, "top_score": 1.0, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 1.0, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "AIME 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "AIME 2025", "benchmark_component_key": "aime_2025", "benchmark_component_name": "Aime 2025", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.aime-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Aime 2025 / Score", "canonical_display_name": "Aime 2025 / Score", "raw_evaluation_name": "llm_stats.aime-2025", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_browsecomp", "benchmark": "BrowseComp", "benchmark_family_key": "llm_stats", "benchmark_family_name": "BrowseComp", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "BrowseComp", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "benchmark_component_key": "browsecomp", "benchmark_component_name": "Browsecomp", "evaluation_name": "Browsecomp", "display_name": "Browsecomp", "canonical_display_name": "Browsecomp", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "BrowseComp", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-mythos-preview", "https://llm-stats.com/benchmarks/browsecomp", "https://api.llm-stats.com/leaderboard/benchmarks/browsecomp" ], "additional_details": { "raw_benchmark_id": "browsecomp", "raw_model_id": "claude-mythos-preview", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_browsecomp_score", "legacy_eval_summary_id": "llm_stats_llm_stats_browsecomp", "evaluation_name": "llm_stats.browsecomp", "display_name": "Browsecomp / Score", "canonical_display_name": "Browsecomp / Score", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.browsecomp.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "BrowseComp is a benchmark comprising 1,266 questions that challenge AI agents to persistently navigate the internet in search of hard-to-find, entangled information. The benchmark measures agents' ability to exercise persistence in information gathering, demonstrate creativity in web navigation, and find concise, verifiable answers. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers.", "metric_id": "llm_stats.browsecomp.score", "metric_name": "BrowseComp score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "browsecomp", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "BrowseComp", "raw_categories": "[\"agents\",\"reasoning\",\"search\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "45" } }, "models_count": 1, "top_score": 0.779, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.779, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "BrowseComp", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "BrowseComp", "benchmark_component_key": "browsecomp", "benchmark_component_name": "Browsecomp", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.browsecomp.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Browsecomp / Score", "canonical_display_name": "Browsecomp / Score", "raw_evaluation_name": "llm_stats.browsecomp", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.779, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_gpqa", "benchmark": "GPQA", "benchmark_family_key": "llm_stats", "benchmark_family_name": "GPQA", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "GPQA", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "benchmark_component_key": "gpqa", "benchmark_component_name": "Gpqa", "evaluation_name": "Gpqa", "display_name": "Gpqa", "canonical_display_name": "Gpqa", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "GPQA", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-mythos-preview", "https://llm-stats.com/benchmarks/gpqa", "https://api.llm-stats.com/leaderboard/benchmarks/gpqa" ], "additional_details": { "raw_benchmark_id": "gpqa", "raw_model_id": "claude-mythos-preview", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_gpqa_score", "legacy_eval_summary_id": "llm_stats_llm_stats_gpqa", "evaluation_name": "llm_stats.gpqa", "display_name": "Gpqa / Score", "canonical_display_name": "Gpqa / Score", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.gpqa.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Questions are Google-proof and extremely difficult, with PhD experts reaching 65% accuracy.", "metric_id": "llm_stats.gpqa.score", "metric_name": "GPQA score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "gpqa", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "GPQA", "raw_categories": "[\"biology\",\"chemistry\",\"general\",\"physics\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "213" } }, "models_count": 1, "top_score": 0.932, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.932, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "GPQA", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "GPQA", "benchmark_component_key": "gpqa", "benchmark_component_name": "Gpqa", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.gpqa.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gpqa / Score", "canonical_display_name": "Gpqa / Score", "raw_evaluation_name": "llm_stats.gpqa", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.932, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_hmmt_2025", "benchmark": "HMMT 2025", "benchmark_family_key": "llm_stats", "benchmark_family_name": "HMMT 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "HMMT 2025", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "benchmark_component_key": "hmmt_2025", "benchmark_component_name": "Hmmt 2025", "evaluation_name": "Hmmt 2025", "display_name": "Hmmt 2025", "canonical_display_name": "Hmmt 2025", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "HMMT 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/deepseek-reasoner", "https://llm-stats.com/benchmarks/hmmt-2025", "https://api.llm-stats.com/leaderboard/benchmarks/hmmt-2025" ], "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_model_id": "deepseek-reasoner", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_hmmt_2025_score", "legacy_eval_summary_id": "llm_stats_llm_stats_hmmt_2025", "evaluation_name": "llm_stats.hmmt-2025", "display_name": "Hmmt 2025 / Score", "canonical_display_name": "Hmmt 2025 / Score", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.hmmt-2025.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "metric_id": "llm_stats.hmmt-2025.score", "metric_name": "HMMT 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "HMMT 2025", "raw_categories": "[\"math\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "32" } }, "models_count": 1, "top_score": 1.0, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 1.0, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "HMMT 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "HMMT 2025", "benchmark_component_key": "hmmt_2025", "benchmark_component_name": "Hmmt 2025", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.hmmt-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Hmmt 2025 / Score", "canonical_display_name": "Hmmt 2025 / Score", "raw_evaluation_name": "llm_stats.hmmt-2025", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "reasoning": [ { "eval_summary_id": "arc_agi", "benchmark": "ARC Prize evaluations leaderboard JSON", "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "evaluation_name": "ARC-AGI", "display_name": "ARC-AGI", "canonical_display_name": "ARC-AGI", "is_summary_score": false, "category": "reasoning", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 4, "metrics_count": 8, "metric_names": [ "Cost per Task", "Score" ], "primary_metric_name": "Cost per Task", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 20, "has_reproducibility_gap_count": 20, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 20, "total_groups": 8, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 20, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 8, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [], "subtasks": [ { "subtask_key": "v1_public_eval", "subtask_name": "v1_Public_Eval", "display_name": "v1_Public_Eval", "metrics": [ { "metric_summary_id": "arc_agi_v1_public_eval_cost_per_task", "legacy_eval_summary_id": "arc_agi_v1_public_eval", "evaluation_name": "v1_Public_Eval", "display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 3, "top_score": 3.2418, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 3.2418, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 4.6384, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 7.7201, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v1_public_eval_score", "legacy_eval_summary_id": "arc_agi_v1_public_eval", "evaluation_name": "v1_Public_Eval", "display_name": "ARC-AGI / v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 3, "top_score": 0.9761, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 0.9761, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.9462, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.9025, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v1_semi_private", "subtask_name": "v1_Semi_Private", "display_name": "v1_Semi_Private", "metrics": [ { "metric_summary_id": "arc_agi_v1_semi_private_cost_per_task", "legacy_eval_summary_id": "arc_agi_v1_semi_private", "evaluation_name": "v1_Semi_Private", "display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 3, "top_score": 3.9774, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 3.9774, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 5.8694, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 11.6542, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v1_semi_private_score", "legacy_eval_summary_id": "arc_agi_v1_semi_private", "evaluation_name": "v1_Semi_Private", "display_name": "ARC-AGI / v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 3, "top_score": 0.905, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 0.905, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.8567, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.8117, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v2_public_eval", "subtask_name": "v2_Public_Eval", "display_name": "v2_Public_Eval", "metrics": [ { "metric_summary_id": "arc_agi_v2_public_eval_cost_per_task", "legacy_eval_summary_id": "arc_agi_v2_public_eval", "evaluation_name": "v2_Public_Eval", "display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 2, "top_score": 9.5162, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 9.5162, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 16.662, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v2_public_eval_score", "legacy_eval_summary_id": "arc_agi_v2_public_eval", "evaluation_name": "v2_Public_Eval", "display_name": "ARC-AGI / v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 2, "top_score": 0.5168, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.5168, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.3792, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v2_semi_private", "subtask_name": "v2_Semi_Private", "display_name": "v2_Semi_Private", "metrics": [ { "metric_summary_id": "arc_agi_v2_semi_private_cost_per_task", "legacy_eval_summary_id": "arc_agi_v2_semi_private", "evaluation_name": "v2_Semi_Private", "display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 2, "top_score": 8.9928, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 8.9928, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 15.721, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v2_semi_private_score", "legacy_eval_summary_id": "arc_agi_v2_semi_private", "evaluation_name": "v2_Semi_Private", "display_name": "ARC-AGI / v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 2, "top_score": 0.5416, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.5416, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.3847, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_arc_agi_v2", "benchmark": "ARC-AGI v2", "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI v2", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI v2", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "benchmark_component_key": "arc_agi_v2", "benchmark_component_name": "Arc Agi V2", "evaluation_name": "Arc Agi V2", "display_name": "Arc Agi V2", "canonical_display_name": "Arc Agi V2", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "ARC-AGI v2", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-opus-4-20250514", "https://llm-stats.com/benchmarks/arc-agi-v2", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi-v2" ], "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_model_id": "claude-opus-4-20250514", "source_role": "aggregator" } }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_arc_agi_v2_score", "legacy_eval_summary_id": "llm_stats_llm_stats_arc_agi_v2", "evaluation_name": "llm_stats.arc-agi-v2", "display_name": "Arc Agi V2 / Score", "canonical_display_name": "Arc Agi V2 / Score", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.arc-agi-v2.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.", "metric_id": "llm_stats.arc-agi-v2.score", "metric_name": "ARC-AGI v2 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI v2", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "multimodal", "raw_verified": "false", "raw_model_count": "15" } }, "models_count": 1, "top_score": 0.542, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.542, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI v2", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI v2", "benchmark_component_key": "arc_agi_v2", "benchmark_component_name": "Arc Agi V2", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi-v2.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi V2 / Score", "canonical_display_name": "Arc Agi V2 / Score", "raw_evaluation_name": "llm_stats.arc-agi-v2", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.542, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_arc_agi", "benchmark": "ARC-AGI", "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "benchmark_component_key": "arc_agi", "benchmark_component_name": "Arc Agi", "evaluation_name": "Arc Agi", "display_name": "Arc Agi", "canonical_display_name": "Arc Agi", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "ARC-AGI", "source_type": "url", "url": [ "https://llm-stats.com/models/longcat-flash-thinking", "https://llm-stats.com/benchmarks/arc-agi", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi" ], "additional_details": { "raw_benchmark_id": "arc-agi", "raw_model_id": "longcat-flash-thinking", "source_role": "aggregator" } }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_arc_agi_score", "legacy_eval_summary_id": "llm_stats_llm_stats_arc_agi", "evaluation_name": "llm_stats.arc-agi", "display_name": "Arc Agi / Score", "canonical_display_name": "Arc Agi / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.arc-agi.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.", "metric_id": "llm_stats.arc-agi.score", "metric_name": "ARC-AGI score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "image", "raw_verified": "false", "raw_model_count": "7" } }, "models_count": 1, "top_score": 0.905, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.905, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI", "benchmark_component_key": "arc_agi", "benchmark_component_name": "Arc Agi", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi / Score", "canonical_display_name": "Arc Agi / Score", "raw_evaluation_name": "llm_stats.arc-agi", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.905, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "hierarchy_by_category": { "knowledge": [ { "eval_summary_id": "apex_v1", "benchmark": "APEX v1", "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "evaluation_name": "APEX v1", "display_name": "APEX v1", "canonical_display_name": "APEX v1", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ] }, "subtasks_count": 3, "metrics_count": 4, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 4, "has_reproducibility_gap_count": 4, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 4, "total_groups": 4, "multi_source_groups": 0, "first_party_only_groups": 4, "source_type_distribution": { "first_party": 4, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 4, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_v1_score", "legacy_eval_summary_id": "apex_v1_apex_v1", "evaluation_name": "apex-v1", "display_name": "APEX v1 / Score", "canonical_display_name": "APEX v1 / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall APEX-v1 mean score (paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.668, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.668, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "consulting", "subtask_name": "Consulting", "display_name": "Consulting", "metrics": [ { "metric_summary_id": "apex_v1_consulting_score", "legacy_eval_summary_id": "apex_v1_consulting", "evaluation_name": "Consulting", "display_name": "APEX v1 / Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.64, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] }, { "subtask_key": "investment_banking", "subtask_name": "Investment Banking", "display_name": "Investment Banking", "metrics": [ { "metric_summary_id": "apex_v1_investment_banking_score", "legacy_eval_summary_id": "apex_v1_investment_banking", "evaluation_name": "Investment Banking", "display_name": "APEX v1 / Investment Banking / Score", "canonical_display_name": "APEX v1 / Investment Banking / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "investment_banking", "slice_name": "Investment Banking", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Investment banking associate score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.64, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Investment Banking / Score", "canonical_display_name": "APEX v1 / Investment Banking / Score", "raw_evaluation_name": "Investment Banking", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] }, { "subtask_key": "medicine_md", "subtask_name": "Medicine (MD)", "display_name": "Medicine (MD)", "metrics": [ { "metric_summary_id": "apex_v1_medicine_md_score", "legacy_eval_summary_id": "apex_v1_medicine_md", "evaluation_name": "Medicine (MD)", "display_name": "APEX v1 / Medicine (MD) / Score", "canonical_display_name": "APEX v1 / Medicine (MD) / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "medicine_md", "slice_name": "Medicine (MD)", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Primary care physician (MD) score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Medicine (MD) Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.65, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT 5.2 Pro", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/GPT 5.2 Pro", "score": 0.65, "evaluation_id": "apex-v1/openai_gpt-5.2-pro/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/apex_v1_openai_gpt_5_2_pro_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "medicine_md", "slice_name": "Medicine (MD)", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Medicine (MD) / Score", "canonical_display_name": "APEX v1 / Medicine (MD) / Score", "raw_evaluation_name": "Medicine (MD)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "other": [ { "eval_summary_id": "llm_stats_aime_2025", "benchmark": "AIME 2025", "benchmark_family_key": "llm_stats", "benchmark_family_name": "AIME 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "AIME 2025", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "benchmark_component_key": "aime_2025", "benchmark_component_name": "Aime 2025", "evaluation_name": "Aime 2025", "display_name": "Aime 2025", "canonical_display_name": "Aime 2025", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-opus-4-6", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "claude-opus-4-6", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_aime_2025_score", "legacy_eval_summary_id": "llm_stats_llm_stats_aime_2025", "evaluation_name": "llm_stats.aime-2025", "display_name": "Aime 2025 / Score", "canonical_display_name": "Aime 2025 / Score", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.aime-2025.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "All 30 problems from the 2025 American Invitational Mathematics Examination (AIME I and AIME II), testing olympiad-level mathematical reasoning with integer answers from 000-999. Used as an AI benchmark to evaluate large language models' ability to solve complex mathematical problems requiring multi-step logical deductions and structured symbolic reasoning.", "metric_id": "llm_stats.aime-2025.score", "metric_name": "AIME 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "aime-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "AIME 2025", "raw_categories": "[\"math\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "107" } }, "models_count": 1, "top_score": 1.0, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 1.0, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "AIME 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "AIME 2025", "benchmark_component_key": "aime_2025", "benchmark_component_name": "Aime 2025", "benchmark_leaf_key": "aime_2025", "benchmark_leaf_name": "Aime 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.aime-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Aime 2025 / Score", "canonical_display_name": "Aime 2025 / Score", "raw_evaluation_name": "llm_stats.aime-2025", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_browsecomp", "benchmark": "BrowseComp", "benchmark_family_key": "llm_stats", "benchmark_family_name": "BrowseComp", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "BrowseComp", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "benchmark_component_key": "browsecomp", "benchmark_component_name": "Browsecomp", "evaluation_name": "Browsecomp", "display_name": "Browsecomp", "canonical_display_name": "Browsecomp", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "BrowseComp", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-mythos-preview", "https://llm-stats.com/benchmarks/browsecomp", "https://api.llm-stats.com/leaderboard/benchmarks/browsecomp" ], "additional_details": { "raw_benchmark_id": "browsecomp", "raw_model_id": "claude-mythos-preview", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_browsecomp_score", "legacy_eval_summary_id": "llm_stats_llm_stats_browsecomp", "evaluation_name": "llm_stats.browsecomp", "display_name": "Browsecomp / Score", "canonical_display_name": "Browsecomp / Score", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.browsecomp.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "BrowseComp is a benchmark comprising 1,266 questions that challenge AI agents to persistently navigate the internet in search of hard-to-find, entangled information. The benchmark measures agents' ability to exercise persistence in information gathering, demonstrate creativity in web navigation, and find concise, verifiable answers. Despite the difficulty of the questions, BrowseComp is simple and easy-to-use, as predicted answers are short and easily verifiable against reference answers.", "metric_id": "llm_stats.browsecomp.score", "metric_name": "BrowseComp score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "browsecomp", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "BrowseComp", "raw_categories": "[\"agents\",\"reasoning\",\"search\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "45" } }, "models_count": 1, "top_score": 0.779, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.779, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "BrowseComp", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "BrowseComp", "benchmark_component_key": "browsecomp", "benchmark_component_name": "Browsecomp", "benchmark_leaf_key": "browsecomp", "benchmark_leaf_name": "Browsecomp", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.browsecomp.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Browsecomp / Score", "canonical_display_name": "Browsecomp / Score", "raw_evaluation_name": "llm_stats.browsecomp", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.779, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_gpqa", "benchmark": "GPQA", "benchmark_family_key": "llm_stats", "benchmark_family_name": "GPQA", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "GPQA", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "benchmark_component_key": "gpqa", "benchmark_component_name": "Gpqa", "evaluation_name": "Gpqa", "display_name": "Gpqa", "canonical_display_name": "Gpqa", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "GPQA", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-mythos-preview", "https://llm-stats.com/benchmarks/gpqa", "https://api.llm-stats.com/leaderboard/benchmarks/gpqa" ], "additional_details": { "raw_benchmark_id": "gpqa", "raw_model_id": "claude-mythos-preview", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_gpqa_score", "legacy_eval_summary_id": "llm_stats_llm_stats_gpqa", "evaluation_name": "llm_stats.gpqa", "display_name": "Gpqa / Score", "canonical_display_name": "Gpqa / Score", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.gpqa.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "A challenging dataset of 448 multiple-choice questions written by domain experts in biology, physics, and chemistry. Questions are Google-proof and extremely difficult, with PhD experts reaching 65% accuracy.", "metric_id": "llm_stats.gpqa.score", "metric_name": "GPQA score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "gpqa", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "GPQA", "raw_categories": "[\"biology\",\"chemistry\",\"general\",\"physics\",\"reasoning\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "213" } }, "models_count": 1, "top_score": 0.932, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.932, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "GPQA", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "GPQA", "benchmark_component_key": "gpqa", "benchmark_component_name": "Gpqa", "benchmark_leaf_key": "gpqa", "benchmark_leaf_name": "Gpqa", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.gpqa.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gpqa / Score", "canonical_display_name": "Gpqa / Score", "raw_evaluation_name": "llm_stats.gpqa", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.932, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_hmmt_2025", "benchmark": "HMMT 2025", "benchmark_family_key": "llm_stats", "benchmark_family_name": "HMMT 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "HMMT 2025", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "benchmark_component_key": "hmmt_2025", "benchmark_component_name": "Hmmt 2025", "evaluation_name": "Hmmt 2025", "display_name": "Hmmt 2025", "canonical_display_name": "Hmmt 2025", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "HMMT 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/deepseek-reasoner", "https://llm-stats.com/benchmarks/hmmt-2025", "https://api.llm-stats.com/leaderboard/benchmarks/hmmt-2025" ], "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_model_id": "deepseek-reasoner", "source_role": "aggregator" } }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_hmmt_2025_score", "legacy_eval_summary_id": "llm_stats_llm_stats_hmmt_2025", "evaluation_name": "llm_stats.hmmt-2025", "display_name": "Hmmt 2025 / Score", "canonical_display_name": "Hmmt 2025 / Score", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.hmmt-2025.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Harvard-MIT Mathematics Tournament 2025 - A prestigious student-organized mathematics competition for high school students featuring two tournaments (November 2025 at MIT and February 2026 at Harvard) with individual tests, team rounds, and guts rounds", "metric_id": "llm_stats.hmmt-2025.score", "metric_name": "HMMT 2025 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "hmmt-2025", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "HMMT 2025", "raw_categories": "[\"math\"]", "raw_modality": "text", "raw_verified": "false", "raw_model_count": "32" } }, "models_count": 1, "top_score": 1.0, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 1.0, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "HMMT 2025", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "HMMT 2025", "benchmark_component_key": "hmmt_2025", "benchmark_component_name": "Hmmt 2025", "benchmark_leaf_key": "hmmt_2025", "benchmark_leaf_name": "Hmmt 2025", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.hmmt-2025.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Hmmt 2025 / Score", "canonical_display_name": "Hmmt 2025 / Score", "raw_evaluation_name": "llm_stats.hmmt-2025", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "reasoning": [ { "eval_summary_id": "arc_agi", "benchmark": "ARC Prize evaluations leaderboard JSON", "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "evaluation_name": "ARC-AGI", "display_name": "ARC-AGI", "canonical_display_name": "ARC-AGI", "is_summary_score": false, "category": "reasoning", "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 4, "metrics_count": 8, "metric_names": [ "Cost per Task", "Score" ], "primary_metric_name": "Cost per Task", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 20, "has_reproducibility_gap_count": 20, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 20, "total_groups": 8, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 20, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 8, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [], "subtasks": [ { "subtask_key": "v1_public_eval", "subtask_name": "v1_Public_Eval", "display_name": "v1_Public_Eval", "metrics": [ { "metric_summary_id": "arc_agi_v1_public_eval_cost_per_task", "legacy_eval_summary_id": "arc_agi_v1_public_eval", "evaluation_name": "v1_Public_Eval", "display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 3, "top_score": 3.2418, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 3.2418, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 4.6384, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 7.7201, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Cost per Task", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v1_public_eval_score", "legacy_eval_summary_id": "arc_agi_v1_public_eval", "evaluation_name": "v1_Public_Eval", "display_name": "ARC-AGI / v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 3, "top_score": 0.9761, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 0.9761, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.9462, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.9025, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_public_eval", "benchmark_component_name": "v1_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_public_eval", "slice_name": "v1_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v1_Public_Eval / Score", "raw_evaluation_name": "v1_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v1_semi_private", "subtask_name": "v1_Semi_Private", "display_name": "v1_Semi_Private", "metrics": [ { "metric_summary_id": "arc_agi_v1_semi_private_cost_per_task", "legacy_eval_summary_id": "arc_agi_v1_semi_private", "evaluation_name": "v1_Semi_Private", "display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 3, "top_score": 3.9774, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 3.9774, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 5.8694, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 11.6542, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Cost per Task", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v1_semi_private_score", "legacy_eval_summary_id": "arc_agi_v1_semi_private", "evaluation_name": "v1_Semi_Private", "display_name": "ARC-AGI / v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 3, "top_score": 0.905, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-xhigh", "developer": "openai", "variant_key": "2025-12-11-xhigh", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-xhigh", "score": 0.905, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-xhigh/1775549757.6016698", "retrieved_timestamp": "1775549757.6016698", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_xhigh_1775549757_6016698.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.8567, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.8117, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v1_semi_private", "benchmark_component_name": "v1_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v1_semi_private", "slice_name": "v1_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v1_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v1_Semi_Private / Score", "raw_evaluation_name": "v1_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v2_public_eval", "subtask_name": "v2_Public_Eval", "display_name": "v2_Public_Eval", "metrics": [ { "metric_summary_id": "arc_agi_v2_public_eval_cost_per_task", "legacy_eval_summary_id": "arc_agi_v2_public_eval", "evaluation_name": "v2_Public_Eval", "display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 2, "top_score": 9.5162, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 9.5162, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 16.662, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Cost per Task", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v2_public_eval_score", "legacy_eval_summary_id": "arc_agi_v2_public_eval", "evaluation_name": "v2_Public_Eval", "display_name": "ARC-AGI / v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 2, "top_score": 0.5168, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.5168, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.3792, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_public_eval", "benchmark_component_name": "v2_Public_Eval", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_public_eval", "slice_name": "v2_Public_Eval", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Public_Eval / Score", "canonical_display_name": "ARC-AGI / v2_Public_Eval / Score", "raw_evaluation_name": "v2_Public_Eval", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] }, { "subtask_key": "v2_semi_private", "subtask_name": "v2_Semi_Private", "display_name": "v2_Semi_Private", "metrics": [ { "metric_summary_id": "arc_agi_v2_semi_private_cost_per_task", "legacy_eval_summary_id": "arc_agi_v2_semi_private", "evaluation_name": "v2_Semi_Private", "display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "lower_is_better": true, "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "metric_config": { "metric_id": "cost_per_task", "metric_name": "Cost per task", "metric_kind": "cost", "metric_unit": "usd", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 77.16309638, "additional_details": { "raw_metric_field": "costPerTask" } }, "models_count": 2, "top_score": 8.9928, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 8.9928, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 15.721, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Cost per Task", "metric_id": "cost_per_task", "metric_key": "cost_per_task", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Cost per Task", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Cost per Task", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "arc_agi_v2_semi_private_score", "legacy_eval_summary_id": "arc_agi_v2_semi_private", "evaluation_name": "v2_Semi_Private", "display_name": "ARC-AGI / v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "lower_is_better": false, "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "metric_id": "score", "metric_name": "ARC score", "metric_kind": "accuracy", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "score" } }, "models_count": 2, "top_score": 0.5416, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-high", "developer": "openai", "variant_key": "2025-12-11-high", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-high", "score": 0.5416, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-high/1775549757.60104", "retrieved_timestamp": "1775549757.60104", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_high_1775549757_60104.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "gpt-5-2-pro-2025-12-11-medium", "developer": "openai", "variant_key": "2025-12-11-medium", "raw_model_id": "openai/gpt-5-2-pro-2025-12-11-medium", "score": 0.3847, "evaluation_id": "arc-agi/openai/gpt-5-2-pro-2025-12-11-medium/1775549757.601366", "retrieved_timestamp": "1775549757.601366", "source_metadata": { "source_name": "ARC Prize leaderboard JSON", "source_type": "documentation", "source_organization_name": "ARC Prize", "source_organization_url": "https://arcprize.org/leaderboard", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://arcprize.org/media/data/leaderboard/evaluations.json", "filtered_to_display_true": "True" } }, "source_data": { "source_type": "url", "dataset_name": "ARC Prize evaluations leaderboard JSON", "url": [ "https://arcprize.org/media/data/leaderboard/evaluations.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/arc_agi_openai_gpt_5_2_pro_2025_12_11_medium_1775549757_601366.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "arc_agi", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "arc_agi", "benchmark_parent_name": "ARC Prize evaluations leaderboard JSON", "benchmark_component_key": "v2_semi_private", "benchmark_component_name": "v2_Semi_Private", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "ARC-AGI", "slice_key": "v2_semi_private", "slice_name": "v2_Semi_Private", "metric_name": "Score", "metric_id": "score", "metric_key": "score", "metric_source": "metric_config", "display_name": "v2_Semi_Private / Score", "canonical_display_name": "ARC-AGI / v2_Semi_Private / Score", "raw_evaluation_name": "v2_Semi_Private", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 2, "metric_names": [ "Cost per Task", "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_arc_agi_v2", "benchmark": "ARC-AGI v2", "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI v2", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI v2", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "benchmark_component_key": "arc_agi_v2", "benchmark_component_name": "Arc Agi V2", "evaluation_name": "Arc Agi V2", "display_name": "Arc Agi V2", "canonical_display_name": "Arc Agi V2", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "ARC-AGI v2", "source_type": "url", "url": [ "https://llm-stats.com/models/claude-opus-4-20250514", "https://llm-stats.com/benchmarks/arc-agi-v2", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi-v2" ], "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_model_id": "claude-opus-4-20250514", "source_role": "aggregator" } }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_arc_agi_v2_score", "legacy_eval_summary_id": "llm_stats_llm_stats_arc_agi_v2", "evaluation_name": "llm_stats.arc-agi-v2", "display_name": "Arc Agi V2 / Score", "canonical_display_name": "Arc Agi V2 / Score", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.arc-agi-v2.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ARC-AGI-2 is an upgraded benchmark for measuring abstract reasoning and problem-solving abilities in AI systems through visual grid transformation tasks. It evaluates fluid intelligence via input-output grid pairs (1x1 to 30x30) using colored cells (0-9), requiring models to identify underlying transformation rules from demonstration examples and apply them to test cases. Designed to be easy for humans but challenging for AI, focusing on core cognitive abilities like spatial reasoning, pattern recognition, and compositional generalization.", "metric_id": "llm_stats.arc-agi-v2.score", "metric_name": "ARC-AGI v2 score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi-v2", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI v2", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "multimodal", "raw_verified": "false", "raw_model_count": "15" } }, "models_count": 1, "top_score": 0.542, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.542, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI v2", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI v2", "benchmark_component_key": "arc_agi_v2", "benchmark_component_name": "Arc Agi V2", "benchmark_leaf_key": "arc_agi_v2", "benchmark_leaf_name": "Arc Agi V2", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi-v2.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi V2 / Score", "canonical_display_name": "Arc Agi V2 / Score", "raw_evaluation_name": "llm_stats.arc-agi-v2", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.542, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "llm_stats_arc_agi", "benchmark": "ARC-AGI", "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "benchmark_component_key": "arc_agi", "benchmark_component_name": "Arc Agi", "evaluation_name": "Arc Agi", "display_name": "Arc Agi", "canonical_display_name": "Arc Agi", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "ARC-AGI", "source_type": "url", "url": [ "https://llm-stats.com/models/longcat-flash-thinking", "https://llm-stats.com/benchmarks/arc-agi", "https://api.llm-stats.com/leaderboard/benchmarks/arc-agi" ], "additional_details": { "raw_benchmark_id": "arc-agi", "raw_model_id": "longcat-flash-thinking", "source_role": "aggregator" } }, "benchmark_card": { "benchmark_details": { "name": "ARC-AGI", "overview": "ARC-AGI (Abstraction and Reasoning Corpus for Artificial General Intelligence) is a benchmark designed to measure a system's ability to generalize and solve novel tasks efficiently, which is considered the essence of intelligence. It consists of 1,000 independent visual reasoning tasks, each following a different logic and designed to be impossible to prepare for in advance.", "data_type": "visual grids", "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2412.04604", "https://huggingface.co/datasets/fchollet/ARC" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To measure generalization on novel tasks, which is considered the essence of intelligence, by evaluating performance on tasks that cannot be prepared for in advance.", "audience": [ "AI researchers working on artificial general intelligence" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ], "limitations": "Not specified", "out_of_scope_uses": [ "Tasks requiring specialized world knowledge (e.g., historical facts)", "Tasks requiring language capabilities" ] }, "data": { "source": "All tasks were created by humans to ensure novelty and diversity.", "size": "1,000 tasks split into four subsets: 400 public training tasks (easy), 400 public evaluation tasks (hard), 100 semi-private evaluation tasks (hard), and 100 private evaluation tasks (hard).", "format": "Each task consists of demonstration pairs (input-output grid examples) and test inputs. A test pair consists of an input grid (rectangular grid with cells containing one of ten values) and an output grid.", "annotation": "Not specified" }, "methodology": { "methods": [ "Test-takers are allowed two attempts per test input", "Models must use demonstration pairs to understand the task and construct output grids for test inputs" ], "metrics": [ "Not specified" ], "calculation": "Not specified", "interpretation": "Not specified", "baseline_results": "The state-of-the-art score on the ARC-AGI private evaluation set increased from 33% to 55.5% as a result of the ARC Prize 2024 competition", "validation": "Public training tasks are designed to expose test-takers to all the Core Knowledge priors needed to solve ARC-AGI tasks" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.limitations", "data.annotation", "methodology.metrics", "methodology.calculation", "methodology.interpretation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T12:17:31.793915", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "artificial general intelligence", "abstraction and reasoning", "visual reasoning" ], "languages": [ "Not specified" ], "tasks": [ "Using demonstration pairs to understand a task's nature", "Constructing output grids corresponding to test inputs" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "llm_stats_arc_agi_score", "legacy_eval_summary_id": "llm_stats_llm_stats_arc_agi", "evaluation_name": "llm_stats.arc-agi", "display_name": "Arc Agi / Score", "canonical_display_name": "Arc Agi / Score", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "llm_stats.arc-agi.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "The Abstraction and Reasoning Corpus for Artificial General Intelligence (ARC-AGI) is a benchmark designed to test general intelligence and abstract reasoning capabilities through visual grid-based transformation tasks. Each task consists of 2-5 demonstration pairs showing input grids transformed into output grids according to underlying rules, with test-takers required to infer these rules and apply them to novel test inputs. The benchmark uses colored grids (up to 30x30) with 10 discrete colors/symbols, designed to measure human-like general fluid intelligence and skill-acquisition efficiency with minimal prior knowledge.", "metric_id": "llm_stats.arc-agi.score", "metric_name": "ARC-AGI score", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_benchmark_id": "arc-agi", "raw_score_field": "score", "bound_strategy": "inferred_proportion", "raw_name": "ARC-AGI", "raw_categories": "[\"spatial_reasoning\",\"vision\",\"reasoning\"]", "raw_modality": "image", "raw_verified": "false", "raw_model_count": "7" } }, "models_count": 1, "top_score": 0.905, "model_results": [ { "model_id": "openai/gpt-5-2-pro", "model_route_id": "openai__gpt-5-2-pro", "model_name": "GPT-5.2 Pro", "developer": "openai", "variant_key": "2025-12-11", "raw_model_id": "openai/gpt-5.2-pro-2025-12-11", "score": 0.905, "evaluation_id": "llm-stats/first_party/openai_gpt-5.2-pro-2025-12-11/1777108064.422824", "retrieved_timestamp": "1777108064.422824", "source_metadata": { "source_name": "LLM Stats API: first_party scores", "source_type": "documentation", "source_organization_name": "LLM Stats", "source_organization_url": "https://llm-stats.com/", "evaluator_relationship": "first_party", "additional_details": { "models_endpoint": "https://api.llm-stats.com/v1/models", "benchmarks_endpoint": "https://api.llm-stats.com/leaderboard/benchmarks", "scores_endpoint": "https://api.llm-stats.com/v1/scores", "scores_endpoint_fallback": "https://api.llm-stats.com/leaderboard/benchmarks/{benchmark_id}", "developer_page_url": "https://llm-stats.com/developer", "attribution_url": "https://llm-stats.com/", "attribution_required": "true", "source_role": "aggregator" } }, "source_data": { "dataset_name": "AIME 2025", "source_type": "url", "url": [ "https://llm-stats.com/models/gpt-5.2-pro-2025-12-11", "https://llm-stats.com/benchmarks/aime-2025", "https://api.llm-stats.com/leaderboard/benchmarks/aime-2025" ], "additional_details": { "raw_benchmark_id": "aime-2025", "raw_model_id": "gpt-5.2-pro-2025-12-11", "source_role": "aggregator" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-2-pro/llm_stats_first_party_openai_gpt_5_2_pro_2025_12_11_1777108064_422824.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "llm_stats", "benchmark_family_name": "ARC-AGI", "benchmark_parent_key": "llm_stats", "benchmark_parent_name": "ARC-AGI", "benchmark_component_key": "arc_agi", "benchmark_component_name": "Arc Agi", "benchmark_leaf_key": "arc_agi", "benchmark_leaf_name": "Arc Agi", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "llm_stats.arc-agi.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Arc Agi / Score", "canonical_display_name": "Arc Agi / Score", "raw_evaluation_name": "llm_stats.arc-agi", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.905, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "total_evaluations": 5, "last_updated": "2026-04-25T09:07:44.422824Z", "categories_covered": [ "knowledge", "other", "reasoning" ], "variants": [ { "variant_key": "default", "variant_label": "Default", "evaluation_count": 1, "raw_model_ids": [ "openai/GPT 5.2 Pro" ], "last_updated": "2026-03-11T20:16:40Z" }, { "variant_key": "2025-12-11-high", "variant_label": "2025-12-11 high", "evaluation_count": 1, "raw_model_ids": [ "openai/gpt-5-2-pro-2025-12-11-high" ], "last_updated": "2026-04-07T08:15:57.601040Z" }, { "variant_key": "2025-12-11-medium", "variant_label": "2025-12-11 medium", "evaluation_count": 1, "raw_model_ids": [ "openai/gpt-5-2-pro-2025-12-11-medium" ], "last_updated": "2026-04-07T08:15:57.601366Z" }, { "variant_key": "2025-12-11-xhigh", "variant_label": "2025-12-11 xhigh", "evaluation_count": 1, "raw_model_ids": [ "openai/gpt-5-2-pro-2025-12-11-xhigh" ], "last_updated": "2026-04-07T08:15:57.601670Z" }, { "variant_key": "2025-12-11", "variant_label": "2025-12-11", "evaluation_count": 1, "raw_model_ids": [ "openai/gpt-5.2-pro-2025-12-11" ], "last_updated": "2026-04-25T09:07:44.422824Z" } ], "reproducibility_summary": { "results_total": 30, "has_reproducibility_gap_count": 30, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 30, "total_groups": 18, "multi_source_groups": 0, "first_party_only_groups": 10, "source_type_distribution": { "first_party": 10, "third_party": 20, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 18, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 } }