{ "model_info": { "name": "Gemini 3 Flash Preview (Non-reasoning)", "id": "google/gemini-3-flash", "developer": "google", "inference_platform": "unknown", "additional_details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_model_name": "Gemini 3 Flash Preview (Non-reasoning)", "raw_model_slug": "gemini-3-flash", "raw_creator_id": "faddc6d9-2c14-445f-9b28-56726f59c793", "raw_creator_name": "Google", "raw_creator_slug": "google", "release_date": "2025-12-17" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash Preview (Non-reasoning)", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash", "model_version": null }, "model_group_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_family_name": "Gemini 3 Flash Preview (Non-reasoning)", "raw_model_ids": [ "google/Gemini 3 Flash", "google/gemini-3-flash" ], "evaluations_by_category": { "other": [ { "schema_version": "0.2.2", "evaluation_id": "ace/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "benchmark": "ace", "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "source_metadata": { "source_name": "Mercor ACE Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "eval_library": { "name": "archipelago", "version": "1.0.0" }, "model_info": { "name": "Gemini 3 Flash", "developer": "google", "id": "google/Gemini 3 Flash", "inference_platform": "unknown", "normalized_id": "google/Gemini 3 Flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "additional_details": { "run_setting": "High" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/ace_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_name": "Gaming", "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "metric_config": { "evaluation_description": "Gaming domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Gaming Score" }, "metric_id": "ace.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.415 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "ace/google_gemini-3-flash/1773260200#gaming#ace_score", "normalized_result": { "benchmark_family_key": "ace", "benchmark_family_name": "ACE", "benchmark_parent_key": "ace", "benchmark_parent_name": "ACE", "benchmark_component_key": "gaming", "benchmark_component_name": "Gaming", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "slice_key": "gaming", "slice_name": "Gaming", "metric_name": "Score", "metric_id": "ace.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gaming / Score", "canonical_display_name": "ACE / Gaming / Score", "raw_evaluation_name": "Gaming", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "ace", "overview": "The ACE benchmark measures evaluation criteria across four specific research domains, providing domain-specific evaluation criteria rather than general performance metrics.", "data_type": "tabular, text", "domains": [ "DIY/home improvement", "food/recipes", "shopping/products", "gaming design" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/Mercor/ACE" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "Not specified", "audience": [ "Not specified" ], "tasks": [ "Not specified" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "Not specified", "size": "Fewer than 1,000 examples", "format": "CSV", "annotation": "Not specified" }, "methodology": { "methods": [ "Not specified" ], "metrics": [ "Overall Score", "Gaming Score", "DIY Score", "Food Score", "Shopping Score" ], "calculation": "No explicit calculation method provided in structured metadata", "interpretation": "Higher scores indicate better performance for all metrics.", "baseline_results": "Evaluation results from Every Eval Ever: GPT 5 achieved 0.5965 (average_across_subjects), GPT 5.2 scored 0.5810, o3 Pro scored 0.5510, GPT 5.1 scored 0.5428, and o3 scored 0.5213. Across 12 evaluated models, scores ranged from 0.332 to 0.5965 with a mean of 0.4643 and standard deviation of 0.0947.", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.goal", "purpose_and_intended_users.audience", "purpose_and_intended_users.tasks", "purpose_and_intended_users.limitations", "purpose_and_intended_users.out_of_scope_uses", "data.source", "data.annotation", "methodology.methods", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:07:34.053021", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "ace" ] }, { "schema_version": "0.2.2", "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "benchmark": "apex-v1", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "eval_library": { "name": "archipelago", "version": "1.0.0" }, "model_info": { "name": "Gemini 3 Flash", "developer": "google", "id": "google/Gemini 3 Flash", "inference_platform": "unknown", "normalized_id": "google/Gemini 3 Flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "additional_details": { "run_setting": "High" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_v1_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_name": "apex-v1", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Overall APEX-v1 mean score across all jobs.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.64, "uncertainty": { "confidence_interval": { "lower": -0.022, "upper": 0.022, "method": "bootstrap" } } }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/google_gemini-3-flash/1773260200#apex_v1#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false } }, { "evaluation_name": "Consulting", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.64 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-v1/google_gemini-3-flash/1773260200#consulting#apex_v1_score", "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "apex_v1" ] }, { "schema_version": "0.2.2", "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "benchmark": "artificial-analysis-llms", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "eval_library": { "name": "Artificial Analysis", "version": "unknown", "additional_details": { "api_reference_url": "https://artificialanalysis.ai/api-reference" } }, "model_info": { "name": "Gemini 3 Flash Preview (Non-reasoning)", "id": "google/gemini-3-flash", "developer": "google", "inference_platform": "unknown", "additional_details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_model_name": "Gemini 3 Flash Preview (Non-reasoning)", "raw_model_slug": "gemini-3-flash", "raw_creator_id": "faddc6d9-2c14-445f-9b28-56726f59c793", "raw_creator_name": "Google", "raw_creator_slug": "google", "release_date": "2025-12-17" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash Preview (Non-reasoning)", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": null, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_artificial_analysis_intelligence_index#artificial_analysis_artificial_analysis_intelligence_index", "evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Artificial Analysis composite intelligence index.", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_name": "Artificial Analysis Intelligence Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.2, "additional_details": { "raw_metric_field": "artificial_analysis_intelligence_index", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 35.0, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.artificial_analysis_intelligence_index" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_intelligence_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Intelligence Index", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_key": "artificial_analysis_intelligence_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_artificial_analysis_coding_index#artificial_analysis_artificial_analysis_coding_index", "evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Artificial Analysis composite coding index.", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_name": "Artificial Analysis Coding Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.3, "additional_details": { "raw_metric_field": "artificial_analysis_coding_index", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 37.8, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.artificial_analysis_coding_index" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_coding_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Coding Index", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_key": "artificial_analysis_coding_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_artificial_analysis_math_index#artificial_analysis_artificial_analysis_math_index", "evaluation_name": "artificial_analysis.artificial_analysis_math_index", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Artificial Analysis composite math index.", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_name": "Artificial Analysis Math Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 99.0, "additional_details": { "raw_metric_field": "artificial_analysis_math_index", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 55.7, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.artificial_analysis_math_index" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_math_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Math Index", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_key": "artificial_analysis_math_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_math_index", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_mmlu_pro#artificial_analysis_mmlu_pro", "evaluation_name": "artificial_analysis.mmlu_pro", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on MMLU-Pro.", "metric_id": "artificial_analysis.mmlu_pro", "metric_name": "MMLU-Pro", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "mmlu_pro", "bound_strategy": "fixed" } }, "score_details": { "score": 0.882, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.mmlu_pro" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_mmlu_pro", "benchmark_component_name": "artificial_analysis.mmlu_pro", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "slice_key": null, "slice_name": null, "metric_name": "MMLU-Pro", "metric_id": "artificial_analysis.mmlu_pro", "metric_key": "mmlu_pro", "metric_source": "metric_config", "display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "canonical_display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "raw_evaluation_name": "artificial_analysis.mmlu_pro", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_gpqa#artificial_analysis_gpqa", "evaluation_name": "artificial_analysis.gpqa", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on GPQA.", "metric_id": "artificial_analysis.gpqa", "metric_name": "GPQA", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "gpqa", "bound_strategy": "fixed" } }, "score_details": { "score": 0.812, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.gpqa" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_gpqa", "benchmark_component_name": "artificial_analysis.gpqa", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "slice_key": null, "slice_name": null, "metric_name": "GPQA", "metric_id": "artificial_analysis.gpqa", "metric_key": "gpqa", "metric_source": "metric_config", "display_name": "artificial_analysis.gpqa / GPQA", "canonical_display_name": "artificial_analysis.gpqa / GPQA", "raw_evaluation_name": "artificial_analysis.gpqa", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_hle#artificial_analysis_hle", "evaluation_name": "artificial_analysis.hle", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on Humanity's Last Exam.", "metric_id": "artificial_analysis.hle", "metric_name": "Humanity's Last Exam", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "hle", "bound_strategy": "fixed" } }, "score_details": { "score": 0.141, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.hle" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_hle", "benchmark_component_name": "artificial_analysis.hle", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "slice_key": null, "slice_name": null, "metric_name": "Hle", "metric_id": "artificial_analysis.hle", "metric_key": "hle", "metric_source": "metric_config", "display_name": "artificial_analysis.hle / Hle", "canonical_display_name": "artificial_analysis.hle / Hle", "raw_evaluation_name": "artificial_analysis.hle", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_livecodebench#artificial_analysis_livecodebench", "evaluation_name": "artificial_analysis.livecodebench", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on LiveCodeBench.", "metric_id": "artificial_analysis.livecodebench", "metric_name": "LiveCodeBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "livecodebench", "bound_strategy": "fixed" } }, "score_details": { "score": 0.797, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.livecodebench" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_livecodebench", "benchmark_component_name": "artificial_analysis.livecodebench", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "slice_key": null, "slice_name": null, "metric_name": "LiveCodeBench", "metric_id": "artificial_analysis.livecodebench", "metric_key": "livecodebench", "metric_source": "metric_config", "display_name": "artificial_analysis.livecodebench / LiveCodeBench", "canonical_display_name": "artificial_analysis.livecodebench / LiveCodeBench", "raw_evaluation_name": "artificial_analysis.livecodebench", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_scicode#artificial_analysis_scicode", "evaluation_name": "artificial_analysis.scicode", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on SciCode.", "metric_id": "artificial_analysis.scicode", "metric_name": "SciCode", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "scicode", "bound_strategy": "fixed" } }, "score_details": { "score": 0.499, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.scicode" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_scicode", "benchmark_component_name": "artificial_analysis.scicode", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "slice_key": null, "slice_name": null, "metric_name": "SciCode", "metric_id": "artificial_analysis.scicode", "metric_key": "scicode", "metric_source": "metric_config", "display_name": "artificial_analysis.scicode / SciCode", "canonical_display_name": "artificial_analysis.scicode / SciCode", "raw_evaluation_name": "artificial_analysis.scicode", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_aime_25#artificial_analysis_aime_25", "evaluation_name": "artificial_analysis.aime_25", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on AIME 2025.", "metric_id": "artificial_analysis.aime_25", "metric_name": "AIME 2025", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "aime_25", "bound_strategy": "fixed" } }, "score_details": { "score": 0.557, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.aime_25" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_aime_25", "benchmark_component_name": "artificial_analysis.aime_25", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "slice_key": null, "slice_name": null, "metric_name": "Aime 25", "metric_id": "artificial_analysis.aime_25", "metric_key": "aime_25", "metric_source": "metric_config", "display_name": "artificial_analysis.aime_25 / Aime 25", "canonical_display_name": "artificial_analysis.aime_25 / Aime 25", "raw_evaluation_name": "artificial_analysis.aime_25", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_ifbench#artificial_analysis_ifbench", "evaluation_name": "artificial_analysis.ifbench", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on IFBench.", "metric_id": "artificial_analysis.ifbench", "metric_name": "IFBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "ifbench", "bound_strategy": "fixed" } }, "score_details": { "score": 0.551, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.ifbench" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_ifbench", "benchmark_component_name": "artificial_analysis.ifbench", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "slice_key": null, "slice_name": null, "metric_name": "IFBench", "metric_id": "artificial_analysis.ifbench", "metric_key": "ifbench", "metric_source": "metric_config", "display_name": "artificial_analysis.ifbench / IFBench", "canonical_display_name": "artificial_analysis.ifbench / IFBench", "raw_evaluation_name": "artificial_analysis.ifbench", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_lcr#artificial_analysis_lcr", "evaluation_name": "artificial_analysis.lcr", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on AA-LCR.", "metric_id": "artificial_analysis.lcr", "metric_name": "AA-LCR", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "lcr", "bound_strategy": "fixed" } }, "score_details": { "score": 0.48, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.lcr" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_lcr", "benchmark_component_name": "artificial_analysis.lcr", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "slice_key": null, "slice_name": null, "metric_name": "Lcr", "metric_id": "artificial_analysis.lcr", "metric_key": "lcr", "metric_source": "metric_config", "display_name": "artificial_analysis.lcr / Lcr", "canonical_display_name": "artificial_analysis.lcr / Lcr", "raw_evaluation_name": "artificial_analysis.lcr", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_terminalbench_hard#artificial_analysis_terminalbench_hard", "evaluation_name": "artificial_analysis.terminalbench_hard", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on Terminal-Bench Hard.", "metric_id": "artificial_analysis.terminalbench_hard", "metric_name": "Terminal-Bench Hard", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "terminalbench_hard", "bound_strategy": "fixed" } }, "score_details": { "score": 0.318, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.terminalbench_hard" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_terminalbench_hard", "benchmark_component_name": "artificial_analysis.terminalbench_hard", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "slice_key": null, "slice_name": null, "metric_name": "Terminalbench Hard", "metric_id": "artificial_analysis.terminalbench_hard", "metric_key": "terminalbench_hard", "metric_source": "metric_config", "display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "canonical_display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "raw_evaluation_name": "artificial_analysis.terminalbench_hard", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_tau2#artificial_analysis_tau2", "evaluation_name": "artificial_analysis.tau2", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Benchmark score on tau2.", "metric_id": "artificial_analysis.tau2", "metric_name": "tau2", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "tau2", "bound_strategy": "fixed" } }, "score_details": { "score": 0.433, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "evaluations.tau2" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_tau2", "benchmark_component_name": "artificial_analysis.tau2", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "slice_key": null, "slice_name": null, "metric_name": "tau2", "metric_id": "artificial_analysis.tau2", "metric_key": "tau2", "metric_source": "metric_config", "display_name": "artificial_analysis.tau2 / tau2", "canonical_display_name": "artificial_analysis.tau2 / tau2", "raw_evaluation_name": "artificial_analysis.tau2", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_price_1m_blended_3_to_1#artificial_analysis_price_1m_blended_3_to_1", "evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Blended price per 1M tokens using a 3:1 input-to-output ratio.", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_name": "Price per 1M tokens (blended 3:1)", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 262.5, "additional_details": { "raw_metric_field": "price_1m_blended_3_to_1", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 1.125, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "pricing.price_1m_blended_3_to_1" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_component_name": "artificial_analysis.price_1m_blended_3_to_1", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Blended 3 To 1", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_key": "price_1m_blended_3_to_1", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "raw_evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_price_1m_input_tokens#artificial_analysis_price_1m_input_tokens", "evaluation_name": "artificial_analysis.price_1m_input_tokens", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Price per 1M input tokens in USD.", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_name": "Price per 1M input tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 150.0, "additional_details": { "raw_metric_field": "price_1m_input_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 0.5, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "pricing.price_1m_input_tokens" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_input_tokens", "benchmark_component_name": "artificial_analysis.price_1m_input_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Input Tokens", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_key": "price_1m_input_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_input_tokens", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_price_1m_output_tokens#artificial_analysis_price_1m_output_tokens", "evaluation_name": "artificial_analysis.price_1m_output_tokens", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Price per 1M output tokens in USD.", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_name": "Price per 1M output tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 600.0, "additional_details": { "raw_metric_field": "price_1m_output_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 3.0, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "pricing.price_1m_output_tokens" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_output_tokens", "benchmark_component_name": "artificial_analysis.price_1m_output_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Output Tokens", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_key": "price_1m_output_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_output_tokens", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_median_output_tokens_per_second#artificial_analysis_median_output_tokens_per_second__parallel_queries_1_0__prompt_length_1000_0", "evaluation_name": "artificial_analysis.median_output_tokens_per_second", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Median output generation speed reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_name": "Median output tokens per second", "metric_kind": "throughput", "metric_unit": "tokens_per_second", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 746.589, "additional_details": { "raw_metric_field": "median_output_tokens_per_second", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 181.859, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "median_output_tokens_per_second" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_component_name": "artificial_analysis.median_output_tokens_per_second", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "slice_key": null, "slice_name": null, "metric_name": "Median output tokens per second", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_key": "median_output_tokens_per_second", "metric_source": "metric_config", "display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "raw_evaluation_name": "artificial_analysis.median_output_tokens_per_second", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_median_time_to_first_token_seconds#artificial_analysis_median_time_to_first_token_seconds__parallel_queries_1_0__prompt_length_1000_0", "evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Median time to first token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_name": "Median time to first token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_token_seconds", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 0.959, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "median_time_to_first_token_seconds" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_component_name": "artificial_analysis.median_time_to_first_token_seconds", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "slice_key": null, "slice_name": null, "metric_name": "Median Time To First Token Seconds", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_key": "median_time_to_first_token_seconds", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "raw_evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "is_summary_score": false } }, { "evaluation_result_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802#artificial_analysis_median_time_to_first_answer_token#artificial_analysis_median_time_to_first_answer_token__parallel_queries_1_0__prompt_length_1000_0", "evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "metric_config": { "evaluation_description": "Median time to first answer token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_name": "Median time to first answer token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_answer_token", "bound_strategy": "observed_max_from_snapshot" } }, "score_details": { "score": 0.959, "details": { "raw_model_id": "783a0ea2-1eef-422a-8c3d-f6d40d943f54", "raw_value_field": "median_time_to_first_answer_token" } }, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_component_name": "artificial_analysis.median_time_to_first_answer_token", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "slice_key": null, "slice_name": null, "metric_name": "Median time to first answer token", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_key": "median_time_to_first_answer_token", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "raw_evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "is_summary_score": false } } ], "benchmark_card": null, "instance_level_data": null, "eval_summary_ids": [ "artificial_analysis_llms_artificial_analysis_aime_25", "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index", "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index", "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index", "artificial_analysis_llms_artificial_analysis_gpqa", "artificial_analysis_llms_artificial_analysis_hle", "artificial_analysis_llms_artificial_analysis_ifbench", "artificial_analysis_llms_artificial_analysis_lcr", "artificial_analysis_llms_artificial_analysis_livecodebench", "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second", "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token", "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds", "artificial_analysis_llms_artificial_analysis_mmlu_pro", "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1", "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens", "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens", "artificial_analysis_llms_artificial_analysis_scicode", "artificial_analysis_llms_artificial_analysis_tau2", "artificial_analysis_llms_artificial_analysis_terminalbench_hard" ] } ], "agentic": [ { "schema_version": "0.2.2", "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "benchmark": "apex-agents", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "eval_library": { "name": "archipelago", "version": "1.0.0" }, "model_info": { "name": "Gemini 3 Flash", "developer": "google", "id": "google/Gemini 3 Flash", "inference_platform": "unknown", "normalized_id": "google/Gemini 3 Flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "additional_details": { "run_setting": "High" } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_name": "Overall", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "score_details": { "score": 0.24, "uncertainty": { "confidence_interval": { "lower": -0.033, "upper": 0.033, "method": "bootstrap" } } }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#overall#pass_at_k__k_1", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false } }, { "evaluation_name": "Overall", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Overall Pass@8 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Pass@8" }, "metric_id": "pass_at_k", "metric_name": "Pass@8", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 8 } }, "score_details": { "score": 0.367, "uncertainty": { "confidence_interval": { "lower": -0.044, "upper": 0.043, "method": "bootstrap" } } }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#overall#pass_at_k__k_8", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false } }, { "evaluation_name": "Overall", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Overall mean rubric score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.395 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#overall#mean_score", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Overall / Mean Score", "canonical_display_name": "APEX Agents / Mean Score", "raw_evaluation_name": "Overall", "is_summary_score": false } }, { "evaluation_name": "Investment Banking", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Investment banking world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "score_details": { "score": 0.267 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#investment_banking#pass_at_k__k_1", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Investment Banking / Pass At K", "canonical_display_name": "APEX Agents / Investment Banking / Pass At K", "raw_evaluation_name": "Investment Banking", "is_summary_score": false } }, { "evaluation_name": "Management Consulting", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Management consulting world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Management Consulting Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "score_details": { "score": 0.193 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#management_consulting#pass_at_k__k_1", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "management_consulting", "benchmark_component_name": "Management Consulting", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "management_consulting", "slice_name": "Management Consulting", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Management Consulting / Pass At K", "canonical_display_name": "APEX Agents / Management Consulting / Pass At K", "raw_evaluation_name": "Management Consulting", "is_summary_score": false } }, { "evaluation_name": "Corporate Law", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Corporate law world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Law Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "score_details": { "score": 0.259 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#corporate_law#pass_at_k__k_1", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_law", "benchmark_component_name": "Corporate Law", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_law", "slice_name": "Corporate Law", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Corporate Law / Pass At K", "canonical_display_name": "APEX Agents / Corporate Law / Pass At K", "raw_evaluation_name": "Corporate Law", "is_summary_score": false } }, { "evaluation_name": "Corporate Lawyer", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "metric_config": { "evaluation_description": "Corporate lawyer world mean score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Lawyer Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "score_details": { "score": 0.524 }, "generation_config": { "additional_details": { "run_setting": "High" } }, "evaluation_result_id": "apex-agents/google_gemini-3-flash/1773260200#corporate_lawyer#mean_score", "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_lawyer", "benchmark_component_name": "Corporate Lawyer", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_lawyer", "slice_name": "Corporate Lawyer", "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Corporate Lawyer / Mean Score", "canonical_display_name": "APEX Agents / Corporate Lawyer / Mean Score", "raw_evaluation_name": "Corporate Lawyer", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "APEX-Agents", "overview": "APEX-Agents (AI Productivity Index for Agents) is a benchmark that evaluates AI agents' ability to execute long-horizon, cross-application professional services tasks created by industry professionals. It features 480 tasks across 33 distinct 'worlds' with realistic work environments containing files and tools, specifically designed for investment banking, management consulting, and corporate law domains.", "data_type": "document, image, text", "domains": [ "investment banking", "management consulting", "corporate law" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/mercor/apex-agents", "https://github.com/Mercor-Intelligence/archipelago" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "To assess whether frontier AI agents can perform complex, long-horizon professional services work that requires reasoning, advanced knowledge, and multi-application tool use.", "audience": [ "Researchers and developers working on AI agents, particularly those interested in evaluating agent performance in professional service domains" ], "tasks": [ "Executing realistic professional work prompts requiring multi-application tool use", "Producing deliverables in investment banking, management consulting, and corporate law domains" ], "limitations": "Agents have substantial headroom to improve with top performers scoring under 25% on the primary metric. There is variance in performance across runs and a risk of self-preference in the judge model. Web search is disabled to maintain reproducibility.", "out_of_scope_uses": [ "Use with web search enabled" ] }, "data": { "source": "The benchmark data was created by industry professionals—investment banking analysts, management consultants, and corporate lawyers—who were assigned to teams, given roles, and tasked with delivering projects over 5-10 days within 33 distinct 'worlds'.", "size": "480 tasks, split evenly across three domains: 160 for investment banking, 160 for management consulting, and 160 for corporate law.", "format": "JSON", "annotation": "Tasks were created by professionals based on their project work. Each task includes a rubric with binary criteria and an expert-created reference output. A baselining study was conducted where experts independently executed 20% of the tasks to check for completeness, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated by executing tasks in provided worlds with all necessary files and tools using a zero-shot approach without web search", "A judge model grades agent outputs against task rubrics by evaluating each binary criterion independently" ], "metrics": [ "Overall Pass@1", "Overall Pass@8", "Overall Mean Score", "Investment Banking Pass@1", "Management Consulting Pass@1", "Corporate Law Pass@1", "Corporate Lawyer Mean Score" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping.", "interpretation": "Pass@1 represents the probability that an agent will pass a task (meet all criteria) on a single attempt. Higher scores indicate better performance.", "baseline_results": "Paper baselines: Gemini 3 Flash (24.0%), GPT-5.2 (23.0%), Claude Opus 4.5 (18.4%), Gemini 3 Pro (18.4%), GPT-OSS-120B (<5%). EEE results: Gemini 3.1 Pro (41.45%), Kimi K2.5 (40.20%), Opus 4.6 (40.00%), GPT 5.1 (37.60%), GPT 5.1 Codex (36.60%).", "validation": "The judge model's performance was validated against a human-labeled ground truth set of 747 criteria labels, achieving 98.5% accuracy." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Decision bias", "description": [ "Decision bias occurs when one group is unfairly advantaged over another due to decisions of the model. This might be caused by biases in the data and also amplified as a result of the model's training." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/decision-bias.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:10:58.110017", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "apex_agents" ] }, { "schema_version": "0.2.2", "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "benchmark": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "eval_library": { "name": "harbor", "version": "unknown" }, "model_info": { "name": "Gemini 3 Flash", "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { "agent_name": "Terminus 2", "agent_organization": "Terminal Bench" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_terminus_2_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-01-07" }, "evaluation_results": [ { "evaluation_name": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "evaluation_timestamp": "2026-01-07", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "score_details": { "score": 51.7, "uncertainty": { "standard_error": { "value": 3.1 }, "num_samples": 435 } }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Terminus 2\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "evaluation_result_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108#terminal_bench_2_0#accuracy", "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "terminal_bench_2_0" ] }, { "schema_version": "0.2.2", "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "benchmark": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "eval_library": { "name": "harbor", "version": "unknown" }, "model_info": { "name": "Gemini 3 Flash", "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { "agent_name": "Gemini CLI", "agent_organization": "Google" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "evaluation_results": [ { "evaluation_name": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "score_details": { "score": 51.0, "uncertainty": { "standard_error": { "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "evaluation_result_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108#terminal_bench_2_0#accuracy", "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "terminal_bench_2_0" ] }, { "schema_version": "0.2.2", "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "benchmark": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "eval_library": { "name": "harbor", "version": "unknown" }, "model_info": { "name": "Gemini 3 Flash", "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { "agent_name": "Junie CLI", "agent_organization": "JetBrains" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_junie_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "evaluation_results": [ { "evaluation_name": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "evaluation_timestamp": "2025-12-23", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "score_details": { "score": 64.3, "uncertainty": { "standard_error": { "value": 2.8 }, "num_samples": 435 } }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Junie CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "evaluation_result_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108#terminal_bench_2_0#accuracy", "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "terminal_bench_2_0" ] }, { "schema_version": "0.2.2", "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "benchmark": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "eval_library": { "name": "harbor", "version": "unknown" }, "model_info": { "name": "Gemini 3 Flash", "id": "google/gemini-3-flash", "developer": "Google", "additional_details": { "agent_name": "Gemini CLI", "agent_organization": "Google" }, "normalized_id": "google/gemini-3-flash", "family_id": "google/gemini-3-flash", "family_slug": "gemini-3-flash", "family_name": "Gemini 3 Flash", "variant_key": "default", "variant_label": "Default", "model_route_id": "google__gemini-3-flash" }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-03-06" }, "evaluation_results": [ { "evaluation_name": "terminal-bench-2.0", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "evaluation_timestamp": "2026-03-06", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "score_details": { "score": 47.4, "uncertainty": { "standard_error": { "value": 3.0 }, "num_samples": 435 } }, "generation_config": { "generation_args": { "execution_command": "harbor run -d terminal-bench@2.0 -a \"Gemini CLI\" -m \"Gemini 3 Flash\" -k 5", "agentic_eval_config": { "available_tools": [ { "name": "terminal", "description": "Full terminal/shell access" } ] }, "max_attempts": 1 } }, "evaluation_result_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108#terminal_bench_2_0#accuracy", "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "instance_level_data": null, "eval_summary_ids": [ "terminal_bench_2_0" ] } ] }, "evaluation_summaries_by_category": { "agentic": [ { "eval_summary_id": "terminal_bench_2_0", "benchmark": "Terminal Bench 2 0", "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "Terminal bench 2 0", "display_name": "Terminal bench 2 0", "canonical_display_name": "Terminal bench 2 0", "is_summary_score": false, "category": "agentic", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Accuracy" ], "primary_metric_name": "Accuracy", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 4, "has_reproducibility_gap_count": 4, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 4, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 4, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "terminal_bench_2_0_accuracy", "legacy_eval_summary_id": "terminal_bench_2_0_terminal_bench_2_0", "evaluation_name": "terminal-bench-2.0", "display_name": "Terminal bench 2 0 / Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "models_count": 4, "top_score": 64.3, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 64.3, "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_junie_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 51.7, "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_terminus_2_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-01-07" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 51.0, "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 47.4, "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-03-06" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 64.3, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "apex_agents", "benchmark": "APEX Agents", "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "evaluation_name": "APEX Agents", "display_name": "APEX Agents", "canonical_display_name": "APEX Agents", "is_summary_score": false, "category": "agentic", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "benchmark_card": { "benchmark_details": { "name": "APEX-Agents", "overview": "APEX-Agents (AI Productivity Index for Agents) is a benchmark that evaluates AI agents' ability to execute long-horizon, cross-application professional services tasks created by industry professionals. It features 480 tasks across 33 distinct 'worlds' with realistic work environments containing files and tools, specifically designed for investment banking, management consulting, and corporate law domains.", "data_type": "document, image, text", "domains": [ "investment banking", "management consulting", "corporate law" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/mercor/apex-agents", "https://github.com/Mercor-Intelligence/archipelago" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "To assess whether frontier AI agents can perform complex, long-horizon professional services work that requires reasoning, advanced knowledge, and multi-application tool use.", "audience": [ "Researchers and developers working on AI agents, particularly those interested in evaluating agent performance in professional service domains" ], "tasks": [ "Executing realistic professional work prompts requiring multi-application tool use", "Producing deliverables in investment banking, management consulting, and corporate law domains" ], "limitations": "Agents have substantial headroom to improve with top performers scoring under 25% on the primary metric. There is variance in performance across runs and a risk of self-preference in the judge model. Web search is disabled to maintain reproducibility.", "out_of_scope_uses": [ "Use with web search enabled" ] }, "data": { "source": "The benchmark data was created by industry professionals—investment banking analysts, management consultants, and corporate lawyers—who were assigned to teams, given roles, and tasked with delivering projects over 5-10 days within 33 distinct 'worlds'.", "size": "480 tasks, split evenly across three domains: 160 for investment banking, 160 for management consulting, and 160 for corporate law.", "format": "JSON", "annotation": "Tasks were created by professionals based on their project work. Each task includes a rubric with binary criteria and an expert-created reference output. A baselining study was conducted where experts independently executed 20% of the tasks to check for completeness, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated by executing tasks in provided worlds with all necessary files and tools using a zero-shot approach without web search", "A judge model grades agent outputs against task rubrics by evaluating each binary criterion independently" ], "metrics": [ "Overall Pass@1", "Overall Pass@8", "Overall Mean Score", "Investment Banking Pass@1", "Management Consulting Pass@1", "Corporate Law Pass@1", "Corporate Lawyer Mean Score" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping.", "interpretation": "Pass@1 represents the probability that an agent will pass a task (meet all criteria) on a single attempt. Higher scores indicate better performance.", "baseline_results": "Paper baselines: Gemini 3 Flash (24.0%), GPT-5.2 (23.0%), Claude Opus 4.5 (18.4%), Gemini 3 Pro (18.4%), GPT-OSS-120B (<5%). EEE results: Gemini 3.1 Pro (41.45%), Kimi K2.5 (40.20%), Opus 4.6 (40.00%), GPT 5.1 (37.60%), GPT 5.1 Codex (36.60%).", "validation": "The judge model's performance was validated against a human-labeled ground truth set of 747 criteria labels, achieving 98.5% accuracy." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Decision bias", "description": [ "Decision bias occurs when one group is unfairly advantaged over another due to decisions of the model. This might be caused by biases in the data and also amplified as a result of the model's training." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/decision-bias.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:10:58.110017", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law" ], "languages": [ "English" ], "tasks": [ "Executing realistic professional work prompts requiring multi-application tool use", "Producing deliverables in investment banking, management consulting, and corporate law domains" ] }, "subtasks_count": 4, "metrics_count": 6, "metric_names": [ "Mean Score", "Pass At K" ], "primary_metric_name": "Mean Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 7, "has_reproducibility_gap_count": 7, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 7, "total_groups": 6, "multi_source_groups": 0, "first_party_only_groups": 6, "source_type_distribution": { "first_party": 7, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 6, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_agents_mean_score", "legacy_eval_summary_id": "apex_agents_overall", "evaluation_name": "Overall", "display_name": "APEX Agents / Mean Score", "canonical_display_name": "APEX Agents / Mean Score", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall mean rubric score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.395, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.395, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Overall / Mean Score", "canonical_display_name": "APEX Agents / Mean Score", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "apex_agents_pass_at_k", "legacy_eval_summary_id": "apex_agents_overall", "evaluation_name": "Overall", "display_name": "APEX Agents / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 2, "top_score": 0.367, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.367, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.24, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "corporate_law", "subtask_name": "Corporate Law", "display_name": "Corporate Law", "metrics": [ { "metric_summary_id": "apex_agents_corporate_law_pass_at_k", "legacy_eval_summary_id": "apex_agents_corporate_law", "evaluation_name": "Corporate Law", "display_name": "APEX Agents / Corporate Law / Pass At K", "canonical_display_name": "APEX Agents / Corporate Law / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_law", "slice_name": "Corporate Law", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Corporate law world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Law Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.259, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.259, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_law", "benchmark_component_name": "Corporate Law", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_law", "slice_name": "Corporate Law", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Corporate Law / Pass At K", "canonical_display_name": "APEX Agents / Corporate Law / Pass At K", "raw_evaluation_name": "Corporate Law", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] }, { "subtask_key": "corporate_lawyer", "subtask_name": "Corporate Lawyer", "display_name": "Corporate Lawyer", "metrics": [ { "metric_summary_id": "apex_agents_corporate_lawyer_mean_score", "legacy_eval_summary_id": "apex_agents_corporate_lawyer", "evaluation_name": "Corporate Lawyer", "display_name": "APEX Agents / Corporate Lawyer / Mean Score", "canonical_display_name": "APEX Agents / Corporate Lawyer / Mean Score", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_lawyer", "slice_name": "Corporate Lawyer", "lower_is_better": false, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Corporate lawyer world mean score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Lawyer Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.524, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.524, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_lawyer", "benchmark_component_name": "Corporate Lawyer", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_lawyer", "slice_name": "Corporate Lawyer", "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Corporate Lawyer / Mean Score", "canonical_display_name": "APEX Agents / Corporate Lawyer / Mean Score", "raw_evaluation_name": "Corporate Lawyer", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Mean Score" ] }, { "subtask_key": "investment_banking", "subtask_name": "Investment Banking", "display_name": "Investment Banking", "metrics": [ { "metric_summary_id": "apex_agents_investment_banking_pass_at_k", "legacy_eval_summary_id": "apex_agents_investment_banking", "evaluation_name": "Investment Banking", "display_name": "APEX Agents / Investment Banking / Pass At K", "canonical_display_name": "APEX Agents / Investment Banking / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "investment_banking", "slice_name": "Investment Banking", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Investment banking world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.267, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.267, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Investment Banking / Pass At K", "canonical_display_name": "APEX Agents / Investment Banking / Pass At K", "raw_evaluation_name": "Investment Banking", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] }, { "subtask_key": "management_consulting", "subtask_name": "Management Consulting", "display_name": "Management Consulting", "metrics": [ { "metric_summary_id": "apex_agents_management_consulting_pass_at_k", "legacy_eval_summary_id": "apex_agents_management_consulting", "evaluation_name": "Management Consulting", "display_name": "APEX Agents / Management Consulting / Pass At K", "canonical_display_name": "APEX Agents / Management Consulting / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "management_consulting", "slice_name": "Management Consulting", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Management Consulting Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.193, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.193, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "management_consulting", "benchmark_component_name": "Management Consulting", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "management_consulting", "slice_name": "Management Consulting", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Management Consulting / Pass At K", "canonical_display_name": "APEX Agents / Management Consulting / Pass At K", "raw_evaluation_name": "Management Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "knowledge": [ { "eval_summary_id": "apex_v1", "benchmark": "APEX v1", "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "evaluation_name": "APEX v1", "display_name": "APEX v1", "canonical_display_name": "APEX v1", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ] }, "subtasks_count": 1, "metrics_count": 2, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 2, "has_reproducibility_gap_count": 2, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 2, "total_groups": 2, "multi_source_groups": 0, "first_party_only_groups": 2, "source_type_distribution": { "first_party": 2, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 2, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_v1_score", "legacy_eval_summary_id": "apex_v1_apex_v1", "evaluation_name": "apex-v1", "display_name": "APEX v1 / Score", "canonical_display_name": "APEX v1 / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall APEX-v1 mean score (paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.64, "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_v1_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "consulting", "subtask_name": "Consulting", "display_name": "Consulting", "metrics": [ { "metric_summary_id": "apex_v1_consulting_score", "legacy_eval_summary_id": "apex_v1_consulting", "evaluation_name": "Consulting", "display_name": "APEX v1 / Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.64, "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_v1_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "other": [ { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "benchmark_component_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_component_name": "artificial_analysis.median_output_tokens_per_second", "evaluation_name": "artificial_analysis.median_output_tokens_per_second", "display_name": "artificial_analysis.median_output_tokens_per_second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median output tokens per second" ], "primary_metric_name": "Median output tokens per second", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second_median_output_tokens_per_second", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second", "evaluation_name": "artificial_analysis.median_output_tokens_per_second", "display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Median output tokens per second", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_key": "median_output_tokens_per_second", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median output generation speed reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_name": "Median output tokens per second", "metric_kind": "throughput", "metric_unit": "tokens_per_second", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 746.589, "additional_details": { "raw_metric_field": "median_output_tokens_per_second", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 181.859, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 181.859, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_component_name": "artificial_analysis.median_output_tokens_per_second", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "slice_key": null, "slice_name": null, "metric_name": "Median output tokens per second", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_key": "median_output_tokens_per_second", "metric_source": "metric_config", "display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "raw_evaluation_name": "artificial_analysis.median_output_tokens_per_second", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 181.859, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "benchmark_component_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_component_name": "artificial_analysis.median_time_to_first_answer_token", "evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "display_name": "artificial_analysis.median_time_to_first_answer_token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median time to first answer token" ], "primary_metric_name": "Median time to first answer token", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token_median_time_to_first_answer_token", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token", "evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Median time to first answer token", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_key": "median_time_to_first_answer_token", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median time to first answer token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_name": "Median time to first answer token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_answer_token", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.959, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.959, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_component_name": "artificial_analysis.median_time_to_first_answer_token", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "slice_key": null, "slice_name": null, "metric_name": "Median time to first answer token", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_key": "median_time_to_first_answer_token", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "raw_evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.959, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "benchmark_component_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_component_name": "artificial_analysis.median_time_to_first_token_seconds", "evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "display_name": "artificial_analysis.median_time_to_first_token_seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median Time To First Token Seconds" ], "primary_metric_name": "Median Time To First Token Seconds", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds_median_time_to_first_token_seconds", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds", "evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Median Time To First Token Seconds", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_key": "median_time_to_first_token_seconds", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median time to first token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_name": "Median time to first token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_token_seconds", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.959, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.959, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_component_name": "artificial_analysis.median_time_to_first_token_seconds", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "slice_key": null, "slice_name": null, "metric_name": "Median Time To First Token Seconds", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_key": "median_time_to_first_token_seconds", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "raw_evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.959, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "benchmark_component_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_component_name": "artificial_analysis.price_1m_blended_3_to_1", "evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "display_name": "artificial_analysis.price_1m_blended_3_to_1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Blended 3 To 1" ], "primary_metric_name": "Price 1m Blended 3 To 1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1_price_1m_blended_3_to_1", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1", "evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Blended 3 To 1", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_key": "price_1m_blended_3_to_1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Blended price per 1M tokens using a 3:1 input-to-output ratio.", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_name": "Price per 1M tokens (blended 3:1)", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 262.5, "additional_details": { "raw_metric_field": "price_1m_blended_3_to_1", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 1.125, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 1.125, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_component_name": "artificial_analysis.price_1m_blended_3_to_1", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Blended 3 To 1", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_key": "price_1m_blended_3_to_1", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "raw_evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.125, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "benchmark_component_key": "artificial_analysis_price_1m_input_tokens", "benchmark_component_name": "artificial_analysis.price_1m_input_tokens", "evaluation_name": "artificial_analysis.price_1m_input_tokens", "display_name": "artificial_analysis.price_1m_input_tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Input Tokens" ], "primary_metric_name": "Price 1m Input Tokens", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens_price_1m_input_tokens", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens", "evaluation_name": "artificial_analysis.price_1m_input_tokens", "display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Input Tokens", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_key": "price_1m_input_tokens", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Price per 1M input tokens in USD.", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_name": "Price per 1M input tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 150.0, "additional_details": { "raw_metric_field": "price_1m_input_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.5, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.5, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_input_tokens", "benchmark_component_name": "artificial_analysis.price_1m_input_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Input Tokens", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_key": "price_1m_input_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_input_tokens", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.5, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "benchmark_component_key": "artificial_analysis_price_1m_output_tokens", "benchmark_component_name": "artificial_analysis.price_1m_output_tokens", "evaluation_name": "artificial_analysis.price_1m_output_tokens", "display_name": "artificial_analysis.price_1m_output_tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Output Tokens" ], "primary_metric_name": "Price 1m Output Tokens", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens_price_1m_output_tokens", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens", "evaluation_name": "artificial_analysis.price_1m_output_tokens", "display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Output Tokens", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_key": "price_1m_output_tokens", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Price per 1M output tokens in USD.", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_name": "Price per 1M output tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 600.0, "additional_details": { "raw_metric_field": "price_1m_output_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 3.0, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 3.0, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_output_tokens", "benchmark_component_name": "artificial_analysis.price_1m_output_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Output Tokens", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_key": "price_1m_output_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_output_tokens", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 3.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_intelligence_index", "evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "display_name": "artificial_analysis.artificial_analysis_intelligence_index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Intelligence Index" ], "primary_metric_name": "Artificial Analysis Intelligence Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index_artificial_analysis_intelligence_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index", "evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Intelligence Index", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_key": "artificial_analysis_intelligence_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite intelligence index.", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_name": "Artificial Analysis Intelligence Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.2, "additional_details": { "raw_metric_field": "artificial_analysis_intelligence_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 35.0, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 35.0, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_intelligence_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Intelligence Index", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_key": "artificial_analysis_intelligence_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 35.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "benchmark_component_key": "artificial_analysis_gpqa", "benchmark_component_name": "artificial_analysis.gpqa", "evaluation_name": "artificial_analysis.gpqa", "display_name": "artificial_analysis.gpqa", "canonical_display_name": "artificial_analysis.gpqa", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "GPQA" ], "primary_metric_name": "GPQA", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa_gpqa", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa", "evaluation_name": "artificial_analysis.gpqa", "display_name": "artificial_analysis.gpqa / GPQA", "canonical_display_name": "artificial_analysis.gpqa / GPQA", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "GPQA", "metric_id": "artificial_analysis.gpqa", "metric_key": "gpqa", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on GPQA.", "metric_id": "artificial_analysis.gpqa", "metric_name": "GPQA", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "gpqa", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.812, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.812, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_gpqa", "benchmark_component_name": "artificial_analysis.gpqa", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "slice_key": null, "slice_name": null, "metric_name": "GPQA", "metric_id": "artificial_analysis.gpqa", "metric_key": "gpqa", "metric_source": "metric_config", "display_name": "artificial_analysis.gpqa / GPQA", "canonical_display_name": "artificial_analysis.gpqa / GPQA", "raw_evaluation_name": "artificial_analysis.gpqa", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.812, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_hle", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "benchmark_component_key": "artificial_analysis_hle", "benchmark_component_name": "artificial_analysis.hle", "evaluation_name": "artificial_analysis.hle", "display_name": "artificial_analysis.hle", "canonical_display_name": "artificial_analysis.hle", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Hle" ], "primary_metric_name": "Hle", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_hle_hle", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_hle", "evaluation_name": "artificial_analysis.hle", "display_name": "artificial_analysis.hle / Hle", "canonical_display_name": "artificial_analysis.hle / Hle", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Hle", "metric_id": "artificial_analysis.hle", "metric_key": "hle", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on Humanity's Last Exam.", "metric_id": "artificial_analysis.hle", "metric_name": "Humanity's Last Exam", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "hle", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.141, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.141, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_hle", "benchmark_component_name": "artificial_analysis.hle", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "slice_key": null, "slice_name": null, "metric_name": "Hle", "metric_id": "artificial_analysis.hle", "metric_key": "hle", "metric_source": "metric_config", "display_name": "artificial_analysis.hle / Hle", "canonical_display_name": "artificial_analysis.hle / Hle", "raw_evaluation_name": "artificial_analysis.hle", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.141, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_scicode", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "benchmark_component_key": "artificial_analysis_scicode", "benchmark_component_name": "artificial_analysis.scicode", "evaluation_name": "artificial_analysis.scicode", "display_name": "artificial_analysis.scicode", "canonical_display_name": "artificial_analysis.scicode", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "SciCode" ], "primary_metric_name": "SciCode", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_scicode_scicode", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_scicode", "evaluation_name": "artificial_analysis.scicode", "display_name": "artificial_analysis.scicode / SciCode", "canonical_display_name": "artificial_analysis.scicode / SciCode", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "SciCode", "metric_id": "artificial_analysis.scicode", "metric_key": "scicode", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on SciCode.", "metric_id": "artificial_analysis.scicode", "metric_name": "SciCode", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "scicode", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.499, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.499, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_scicode", "benchmark_component_name": "artificial_analysis.scicode", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "slice_key": null, "slice_name": null, "metric_name": "SciCode", "metric_id": "artificial_analysis.scicode", "metric_key": "scicode", "metric_source": "metric_config", "display_name": "artificial_analysis.scicode / SciCode", "canonical_display_name": "artificial_analysis.scicode / SciCode", "raw_evaluation_name": "artificial_analysis.scicode", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.499, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_coding_index", "evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "display_name": "artificial_analysis.artificial_analysis_coding_index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Coding Index" ], "primary_metric_name": "Artificial Analysis Coding Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index_artificial_analysis_coding_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index", "evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Coding Index", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_key": "artificial_analysis_coding_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite coding index.", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_name": "Artificial Analysis Coding Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.3, "additional_details": { "raw_metric_field": "artificial_analysis_coding_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 37.8, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 37.8, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_coding_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Coding Index", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_key": "artificial_analysis_coding_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 37.8, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "benchmark_component_key": "artificial_analysis_ifbench", "benchmark_component_name": "artificial_analysis.ifbench", "evaluation_name": "artificial_analysis.ifbench", "display_name": "artificial_analysis.ifbench", "canonical_display_name": "artificial_analysis.ifbench", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "IFBench" ], "primary_metric_name": "IFBench", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench_ifbench", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench", "evaluation_name": "artificial_analysis.ifbench", "display_name": "artificial_analysis.ifbench / IFBench", "canonical_display_name": "artificial_analysis.ifbench / IFBench", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "IFBench", "metric_id": "artificial_analysis.ifbench", "metric_key": "ifbench", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on IFBench.", "metric_id": "artificial_analysis.ifbench", "metric_name": "IFBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "ifbench", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.551, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.551, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_ifbench", "benchmark_component_name": "artificial_analysis.ifbench", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "slice_key": null, "slice_name": null, "metric_name": "IFBench", "metric_id": "artificial_analysis.ifbench", "metric_key": "ifbench", "metric_source": "metric_config", "display_name": "artificial_analysis.ifbench / IFBench", "canonical_display_name": "artificial_analysis.ifbench / IFBench", "raw_evaluation_name": "artificial_analysis.ifbench", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.551, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_lcr", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "benchmark_component_key": "artificial_analysis_lcr", "benchmark_component_name": "artificial_analysis.lcr", "evaluation_name": "artificial_analysis.lcr", "display_name": "artificial_analysis.lcr", "canonical_display_name": "artificial_analysis.lcr", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Lcr" ], "primary_metric_name": "Lcr", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_lcr_lcr", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_lcr", "evaluation_name": "artificial_analysis.lcr", "display_name": "artificial_analysis.lcr / Lcr", "canonical_display_name": "artificial_analysis.lcr / Lcr", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Lcr", "metric_id": "artificial_analysis.lcr", "metric_key": "lcr", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on AA-LCR.", "metric_id": "artificial_analysis.lcr", "metric_name": "AA-LCR", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "lcr", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.48, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.48, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_lcr", "benchmark_component_name": "artificial_analysis.lcr", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "slice_key": null, "slice_name": null, "metric_name": "Lcr", "metric_id": "artificial_analysis.lcr", "metric_key": "lcr", "metric_source": "metric_config", "display_name": "artificial_analysis.lcr / Lcr", "canonical_display_name": "artificial_analysis.lcr / Lcr", "raw_evaluation_name": "artificial_analysis.lcr", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.48, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_tau2", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "benchmark_component_key": "artificial_analysis_tau2", "benchmark_component_name": "artificial_analysis.tau2", "evaluation_name": "artificial_analysis.tau2", "display_name": "artificial_analysis.tau2", "canonical_display_name": "artificial_analysis.tau2", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "tau2" ], "primary_metric_name": "tau2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_tau2_tau2", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_tau2", "evaluation_name": "artificial_analysis.tau2", "display_name": "artificial_analysis.tau2 / tau2", "canonical_display_name": "artificial_analysis.tau2 / tau2", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "tau2", "metric_id": "artificial_analysis.tau2", "metric_key": "tau2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on tau2.", "metric_id": "artificial_analysis.tau2", "metric_name": "tau2", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "tau2", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.433, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.433, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_tau2", "benchmark_component_name": "artificial_analysis.tau2", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "slice_key": null, "slice_name": null, "metric_name": "tau2", "metric_id": "artificial_analysis.tau2", "metric_key": "tau2", "metric_source": "metric_config", "display_name": "artificial_analysis.tau2 / tau2", "canonical_display_name": "artificial_analysis.tau2 / tau2", "raw_evaluation_name": "artificial_analysis.tau2", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.433, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "benchmark_component_key": "artificial_analysis_terminalbench_hard", "benchmark_component_name": "artificial_analysis.terminalbench_hard", "evaluation_name": "artificial_analysis.terminalbench_hard", "display_name": "artificial_analysis.terminalbench_hard", "canonical_display_name": "artificial_analysis.terminalbench_hard", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Terminalbench Hard" ], "primary_metric_name": "Terminalbench Hard", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard_terminalbench_hard", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard", "evaluation_name": "artificial_analysis.terminalbench_hard", "display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "canonical_display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Terminalbench Hard", "metric_id": "artificial_analysis.terminalbench_hard", "metric_key": "terminalbench_hard", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on Terminal-Bench Hard.", "metric_id": "artificial_analysis.terminalbench_hard", "metric_name": "Terminal-Bench Hard", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "terminalbench_hard", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.318, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.318, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_terminalbench_hard", "benchmark_component_name": "artificial_analysis.terminalbench_hard", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "slice_key": null, "slice_name": null, "metric_name": "Terminalbench Hard", "metric_id": "artificial_analysis.terminalbench_hard", "metric_key": "terminalbench_hard", "metric_source": "metric_config", "display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "canonical_display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "raw_evaluation_name": "artificial_analysis.terminalbench_hard", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.318, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "benchmark_component_key": "artificial_analysis_mmlu_pro", "benchmark_component_name": "artificial_analysis.mmlu_pro", "evaluation_name": "artificial_analysis.mmlu_pro", "display_name": "artificial_analysis.mmlu_pro", "canonical_display_name": "artificial_analysis.mmlu_pro", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "MMLU-Pro" ], "primary_metric_name": "MMLU-Pro", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro_mmlu_pro", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro", "evaluation_name": "artificial_analysis.mmlu_pro", "display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "canonical_display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "MMLU-Pro", "metric_id": "artificial_analysis.mmlu_pro", "metric_key": "mmlu_pro", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on MMLU-Pro.", "metric_id": "artificial_analysis.mmlu_pro", "metric_name": "MMLU-Pro", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "mmlu_pro", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.882, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.882, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_mmlu_pro", "benchmark_component_name": "artificial_analysis.mmlu_pro", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "slice_key": null, "slice_name": null, "metric_name": "MMLU-Pro", "metric_id": "artificial_analysis.mmlu_pro", "metric_key": "mmlu_pro", "metric_source": "metric_config", "display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "canonical_display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "raw_evaluation_name": "artificial_analysis.mmlu_pro", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.882, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "benchmark_component_key": "artificial_analysis_livecodebench", "benchmark_component_name": "artificial_analysis.livecodebench", "evaluation_name": "artificial_analysis.livecodebench", "display_name": "artificial_analysis.livecodebench", "canonical_display_name": "artificial_analysis.livecodebench", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "LiveCodeBench" ], "primary_metric_name": "LiveCodeBench", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench_livecodebench", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench", "evaluation_name": "artificial_analysis.livecodebench", "display_name": "artificial_analysis.livecodebench / LiveCodeBench", "canonical_display_name": "artificial_analysis.livecodebench / LiveCodeBench", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "LiveCodeBench", "metric_id": "artificial_analysis.livecodebench", "metric_key": "livecodebench", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on LiveCodeBench.", "metric_id": "artificial_analysis.livecodebench", "metric_name": "LiveCodeBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "livecodebench", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.797, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.797, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_livecodebench", "benchmark_component_name": "artificial_analysis.livecodebench", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "slice_key": null, "slice_name": null, "metric_name": "LiveCodeBench", "metric_id": "artificial_analysis.livecodebench", "metric_key": "livecodebench", "metric_source": "metric_config", "display_name": "artificial_analysis.livecodebench / LiveCodeBench", "canonical_display_name": "artificial_analysis.livecodebench / LiveCodeBench", "raw_evaluation_name": "artificial_analysis.livecodebench", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.797, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "benchmark_component_key": "artificial_analysis_aime_25", "benchmark_component_name": "artificial_analysis.aime_25", "evaluation_name": "artificial_analysis.aime_25", "display_name": "artificial_analysis.aime_25", "canonical_display_name": "artificial_analysis.aime_25", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Aime 25" ], "primary_metric_name": "Aime 25", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25_aime_25", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25", "evaluation_name": "artificial_analysis.aime_25", "display_name": "artificial_analysis.aime_25 / Aime 25", "canonical_display_name": "artificial_analysis.aime_25 / Aime 25", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Aime 25", "metric_id": "artificial_analysis.aime_25", "metric_key": "aime_25", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on AIME 2025.", "metric_id": "artificial_analysis.aime_25", "metric_name": "AIME 2025", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "aime_25", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.557, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.557, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_aime_25", "benchmark_component_name": "artificial_analysis.aime_25", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "slice_key": null, "slice_name": null, "metric_name": "Aime 25", "metric_id": "artificial_analysis.aime_25", "metric_key": "aime_25", "metric_source": "metric_config", "display_name": "artificial_analysis.aime_25 / Aime 25", "canonical_display_name": "artificial_analysis.aime_25 / Aime 25", "raw_evaluation_name": "artificial_analysis.aime_25", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.557, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_math_index", "evaluation_name": "artificial_analysis.artificial_analysis_math_index", "display_name": "artificial_analysis.artificial_analysis_math_index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Math Index" ], "primary_metric_name": "Artificial Analysis Math Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index_artificial_analysis_math_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index", "evaluation_name": "artificial_analysis.artificial_analysis_math_index", "display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Math Index", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_key": "artificial_analysis_math_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite math index.", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_name": "Artificial Analysis Math Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 99.0, "additional_details": { "raw_metric_field": "artificial_analysis_math_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 55.7, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 55.7, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_math_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Math Index", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_key": "artificial_analysis_math_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_math_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 55.7, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "ace", "benchmark": "ACE", "benchmark_family_key": "ace", "benchmark_family_name": "ACE", "benchmark_parent_key": "ace", "benchmark_parent_name": "ACE", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "ACE", "display_name": "ACE", "canonical_display_name": "ACE", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "benchmark_card": { "benchmark_details": { "name": "ace", "overview": "The ACE benchmark measures evaluation criteria across four specific research domains, providing domain-specific evaluation criteria rather than general performance metrics.", "data_type": "tabular, text", "domains": [ "DIY/home improvement", "food/recipes", "shopping/products", "gaming design" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/Mercor/ACE" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "Not specified", "audience": [ "Not specified" ], "tasks": [ "Not specified" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "Not specified", "size": "Fewer than 1,000 examples", "format": "CSV", "annotation": "Not specified" }, "methodology": { "methods": [ "Not specified" ], "metrics": [ "Overall Score", "Gaming Score", "DIY Score", "Food Score", "Shopping Score" ], "calculation": "No explicit calculation method provided in structured metadata", "interpretation": "Higher scores indicate better performance for all metrics.", "baseline_results": "Evaluation results from Every Eval Ever: GPT 5 achieved 0.5965 (average_across_subjects), GPT 5.2 scored 0.5810, o3 Pro scored 0.5510, GPT 5.1 scored 0.5428, and o3 scored 0.5213. Across 12 evaluated models, scores ranged from 0.332 to 0.5965 with a mean of 0.4643 and standard deviation of 0.0947.", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.goal", "purpose_and_intended_users.audience", "purpose_and_intended_users.tasks", "purpose_and_intended_users.limitations", "purpose_and_intended_users.out_of_scope_uses", "data.source", "data.annotation", "methodology.methods", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:07:34.053021", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "DIY/home improvement", "food/recipes", "shopping/products", "gaming design" ], "languages": [ "Not specified" ], "tasks": [ "Not specified" ] }, "subtasks_count": 1, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [], "subtasks": [ { "subtask_key": "gaming", "subtask_name": "Gaming", "display_name": "Gaming", "metrics": [ { "metric_summary_id": "ace_gaming_score", "legacy_eval_summary_id": "ace_gaming", "evaluation_name": "Gaming", "display_name": "ACE / Gaming / Score", "canonical_display_name": "ACE / Gaming / Score", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "slice_key": "gaming", "slice_name": "Gaming", "lower_is_better": false, "metric_name": "Score", "metric_id": "ace.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Gaming domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Gaming Score" }, "metric_id": "ace.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.415, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.415, "evaluation_id": "ace/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor ACE Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/ace_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "ace", "benchmark_family_name": "ACE", "benchmark_parent_key": "ace", "benchmark_parent_name": "ACE", "benchmark_component_key": "gaming", "benchmark_component_name": "Gaming", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "slice_key": "gaming", "slice_name": "Gaming", "metric_name": "Score", "metric_id": "ace.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gaming / Score", "canonical_display_name": "ACE / Gaming / Score", "raw_evaluation_name": "Gaming", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "hierarchy_by_category": { "agentic": [ { "eval_summary_id": "terminal_bench_2_0", "benchmark": "Terminal Bench 2 0", "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "Terminal bench 2 0", "display_name": "Terminal bench 2 0", "canonical_display_name": "Terminal bench 2 0", "is_summary_score": false, "category": "agentic", "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "benchmark_card": { "benchmark_details": { "name": "terminal-bench-2.0", "overview": "Terminal-bench-2.0 is a benchmark that measures the capabilities of agents and language models to perform valuable work in containerized terminal environments. It is distinctive for its focus on testing practical tasks within terminal environments and is used by virtually all frontier labs.", "benchmark_type": "single", "appears_in": [], "data_type": "Not specified", "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/harborframework/terminal-bench-2.0", "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "purpose_and_intended_users": { "goal": "To evaluate and measure the capabilities of AI agents and language models working in terminal environments.", "audience": [ "AI researchers", "Language model developers" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ], "limitations": "No limitations information found in provided facts", "out_of_scope_uses": "No out-of-scope uses information found in provided facts" }, "data": { "source": "The data originates from the Terminal-Bench 2.0 GitHub repository. The specific methodology for data collection is not described.", "size": "Fewer than 1,000 examples", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "Terminal-bench-2.0 evaluates AI models by running them on terminal-based tasks within containerized environments using the Harbor framework. The benchmark includes an 'oracle' agent for running solutions." ], "metrics": [ "Task resolution accuracy across 87 terminal tasks with 5 trials each" ], "calculation": "The benchmark measures task resolution accuracy across 87 terminal tasks with 5 trials each. The score is continuous and lower values are not better.", "interpretation": "Higher scores indicate better performance. The benchmark evaluates models on their ability to complete terminal tasks accurately.", "baseline_results": "Based on 11 model evaluations from Every Eval Ever: mean terminal-bench-2.0 = 43.7200 (std = 17.5255). Top performers: Claude Opus 4.6 (74.7000), Claude Opus 4.6 (71.9000), Claude Opus 4.6 (69.9000).", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "methodology.metrics": "[Possible Hallucination], no supporting evidence found in source material", "methodology.calculation": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "benchmark_details.data_type", "benchmark_details.similar_benchmarks", "data.format", "data.annotation", "methodology.baseline_results", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T02:11:51.373429", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "benchmark", "agents", "terminal", "code", "evaluation", "harbor" ], "languages": [ "English" ], "tasks": [ "Text generation", "Protein assembly", "Debugging async code", "Resolving security vulnerabilities" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Accuracy" ], "primary_metric_name": "Accuracy", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 4, "has_reproducibility_gap_count": 4, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 4, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 4, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "terminal_bench_2_0_accuracy", "legacy_eval_summary_id": "terminal_bench_2_0_terminal_bench_2_0", "evaluation_name": "terminal-bench-2.0", "display_name": "Terminal bench 2 0 / Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Task resolution accuracy across 87 terminal tasks with 5 trials each", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 100.0, "metric_id": "accuracy", "metric_name": "Task Resolution Accuracy", "metric_kind": "accuracy", "metric_unit": "percentage" }, "models_count": 4, "top_score": 64.3, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 64.3, "evaluation_id": "terminal-bench-2.0/junie-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_junie_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 51.7, "evaluation_id": "terminal-bench-2.0/terminus-2__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_terminus_2_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-01-07" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 51.0, "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2025-12-23" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "Google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 47.4, "evaluation_id": "terminal-bench-2.0/gemini-cli__gemini-3-flash/1773776901.772108", "retrieved_timestamp": "1773776901.772108", "source_metadata": { "source_name": "Terminal-Bench 2.0", "source_type": "documentation", "source_organization_name": "Terminal-Bench", "source_organization_url": "https://www.tbench.ai", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "terminal-bench-2.0", "source_type": "url", "url": [ "https://www.tbench.ai/leaderboard/terminal-bench/2.0" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/terminal_bench_2_0_gemini_cli_gemini_3_flash_1773776901_772108.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": { "evaluation_timestamp": "2026-03-06" }, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "terminal_bench_2_0", "benchmark_family_name": "Terminal bench 2 0", "benchmark_parent_key": "terminal_bench_2_0", "benchmark_parent_name": "Terminal Bench 2 0", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "terminal_bench_2_0", "benchmark_leaf_name": "Terminal bench 2 0", "slice_key": null, "slice_name": null, "metric_name": "Accuracy", "metric_id": "accuracy", "metric_key": "accuracy", "metric_source": "metric_config", "display_name": "Accuracy", "canonical_display_name": "Terminal bench 2 0 / Accuracy", "raw_evaluation_name": "terminal-bench-2.0", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 64.3, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "apex_agents", "benchmark": "APEX Agents", "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "evaluation_name": "APEX Agents", "display_name": "APEX Agents", "canonical_display_name": "APEX Agents", "is_summary_score": false, "category": "agentic", "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "benchmark_card": { "benchmark_details": { "name": "APEX-Agents", "overview": "APEX-Agents (AI Productivity Index for Agents) is a benchmark that evaluates AI agents' ability to execute long-horizon, cross-application professional services tasks created by industry professionals. It features 480 tasks across 33 distinct 'worlds' with realistic work environments containing files and tools, specifically designed for investment banking, management consulting, and corporate law domains.", "data_type": "document, image, text", "domains": [ "investment banking", "management consulting", "corporate law" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/mercor/apex-agents", "https://github.com/Mercor-Intelligence/archipelago" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "To assess whether frontier AI agents can perform complex, long-horizon professional services work that requires reasoning, advanced knowledge, and multi-application tool use.", "audience": [ "Researchers and developers working on AI agents, particularly those interested in evaluating agent performance in professional service domains" ], "tasks": [ "Executing realistic professional work prompts requiring multi-application tool use", "Producing deliverables in investment banking, management consulting, and corporate law domains" ], "limitations": "Agents have substantial headroom to improve with top performers scoring under 25% on the primary metric. There is variance in performance across runs and a risk of self-preference in the judge model. Web search is disabled to maintain reproducibility.", "out_of_scope_uses": [ "Use with web search enabled" ] }, "data": { "source": "The benchmark data was created by industry professionals—investment banking analysts, management consultants, and corporate lawyers—who were assigned to teams, given roles, and tasked with delivering projects over 5-10 days within 33 distinct 'worlds'.", "size": "480 tasks, split evenly across three domains: 160 for investment banking, 160 for management consulting, and 160 for corporate law.", "format": "JSON", "annotation": "Tasks were created by professionals based on their project work. Each task includes a rubric with binary criteria and an expert-created reference output. A baselining study was conducted where experts independently executed 20% of the tasks to check for completeness, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated by executing tasks in provided worlds with all necessary files and tools using a zero-shot approach without web search", "A judge model grades agent outputs against task rubrics by evaluating each binary criterion independently" ], "metrics": [ "Overall Pass@1", "Overall Pass@8", "Overall Mean Score", "Investment Banking Pass@1", "Management Consulting Pass@1", "Corporate Law Pass@1", "Corporate Lawyer Mean Score" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping.", "interpretation": "Pass@1 represents the probability that an agent will pass a task (meet all criteria) on a single attempt. Higher scores indicate better performance.", "baseline_results": "Paper baselines: Gemini 3 Flash (24.0%), GPT-5.2 (23.0%), Claude Opus 4.5 (18.4%), Gemini 3 Pro (18.4%), GPT-OSS-120B (<5%). EEE results: Gemini 3.1 Pro (41.45%), Kimi K2.5 (40.20%), Opus 4.6 (40.00%), GPT 5.1 (37.60%), GPT 5.1 Codex (36.60%).", "validation": "The judge model's performance was validated against a human-labeled ground truth set of 747 criteria labels, achieving 98.5% accuracy." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Decision bias", "description": [ "Decision bias occurs when one group is unfairly advantaged over another due to decisions of the model. This might be caused by biases in the data and also amplified as a result of the model's training." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/decision-bias.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:10:58.110017", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law" ], "languages": [ "English" ], "tasks": [ "Executing realistic professional work prompts requiring multi-application tool use", "Producing deliverables in investment banking, management consulting, and corporate law domains" ] }, "subtasks_count": 4, "metrics_count": 6, "metric_names": [ "Mean Score", "Pass At K" ], "primary_metric_name": "Mean Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 7, "has_reproducibility_gap_count": 7, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 7, "total_groups": 6, "multi_source_groups": 0, "first_party_only_groups": 6, "source_type_distribution": { "first_party": 7, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 6, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_agents_mean_score", "legacy_eval_summary_id": "apex_agents_overall", "evaluation_name": "Overall", "display_name": "APEX Agents / Mean Score", "canonical_display_name": "APEX Agents / Mean Score", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall mean rubric score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.395, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.395, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Overall / Mean Score", "canonical_display_name": "APEX Agents / Mean Score", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] }, { "metric_summary_id": "apex_agents_pass_at_k", "legacy_eval_summary_id": "apex_agents_overall", "evaluation_name": "Overall", "display_name": "APEX Agents / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall Pass@1 (dataset card / paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 2, "top_score": 0.367, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.367, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.24, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "overall", "benchmark_component_name": "Overall", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": null, "slice_name": null, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Overall / Pass At K", "canonical_display_name": "APEX Agents / Pass At K", "raw_evaluation_name": "Overall", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "corporate_law", "subtask_name": "Corporate Law", "display_name": "Corporate Law", "metrics": [ { "metric_summary_id": "apex_agents_corporate_law_pass_at_k", "legacy_eval_summary_id": "apex_agents_corporate_law", "evaluation_name": "Corporate Law", "display_name": "APEX Agents / Corporate Law / Pass At K", "canonical_display_name": "APEX Agents / Corporate Law / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_law", "slice_name": "Corporate Law", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Corporate law world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Law Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.259, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.259, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_law", "benchmark_component_name": "Corporate Law", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_law", "slice_name": "Corporate Law", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Corporate Law / Pass At K", "canonical_display_name": "APEX Agents / Corporate Law / Pass At K", "raw_evaluation_name": "Corporate Law", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] }, { "subtask_key": "corporate_lawyer", "subtask_name": "Corporate Lawyer", "display_name": "Corporate Lawyer", "metrics": [ { "metric_summary_id": "apex_agents_corporate_lawyer_mean_score", "legacy_eval_summary_id": "apex_agents_corporate_lawyer", "evaluation_name": "Corporate Lawyer", "display_name": "APEX Agents / Corporate Lawyer / Mean Score", "canonical_display_name": "APEX Agents / Corporate Lawyer / Mean Score", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_lawyer", "slice_name": "Corporate Lawyer", "lower_is_better": false, "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Corporate lawyer world mean score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Corporate Lawyer Mean Score" }, "metric_id": "mean_score", "metric_name": "Mean Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.524, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.524, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "corporate_lawyer", "benchmark_component_name": "Corporate Lawyer", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "corporate_lawyer", "slice_name": "Corporate Lawyer", "metric_name": "Mean Score", "metric_id": "mean_score", "metric_key": "mean_score", "metric_source": "metric_config", "display_name": "Corporate Lawyer / Mean Score", "canonical_display_name": "APEX Agents / Corporate Lawyer / Mean Score", "raw_evaluation_name": "Corporate Lawyer", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Mean Score" ] }, { "subtask_key": "investment_banking", "subtask_name": "Investment Banking", "display_name": "Investment Banking", "metrics": [ { "metric_summary_id": "apex_agents_investment_banking_pass_at_k", "legacy_eval_summary_id": "apex_agents_investment_banking", "evaluation_name": "Investment Banking", "display_name": "APEX Agents / Investment Banking / Pass At K", "canonical_display_name": "APEX Agents / Investment Banking / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "investment_banking", "slice_name": "Investment Banking", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Investment banking world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Investment Banking Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.267, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.267, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "investment_banking", "benchmark_component_name": "Investment Banking", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "investment_banking", "slice_name": "Investment Banking", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Investment Banking / Pass At K", "canonical_display_name": "APEX Agents / Investment Banking / Pass At K", "raw_evaluation_name": "Investment Banking", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] }, { "subtask_key": "management_consulting", "subtask_name": "Management Consulting", "display_name": "Management Consulting", "metrics": [ { "metric_summary_id": "apex_agents_management_consulting_pass_at_k", "legacy_eval_summary_id": "apex_agents_management_consulting", "evaluation_name": "Management Consulting", "display_name": "APEX Agents / Management Consulting / Pass At K", "canonical_display_name": "APEX Agents / Management Consulting / Pass At K", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "management_consulting", "slice_name": "Management Consulting", "lower_is_better": false, "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting world Pass@1.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Management Consulting Pass@1" }, "metric_id": "pass_at_k", "metric_name": "Pass@1", "metric_kind": "pass_rate", "metric_unit": "proportion", "metric_parameters": { "k": 1 } }, "models_count": 1, "top_score": 0.193, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.193, "evaluation_id": "apex-agents/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-Agents Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-agents", "source_type": "hf_dataset", "hf_repo": "mercor/apex-agents" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_agents_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_agents", "benchmark_family_name": "APEX Agents", "benchmark_parent_key": "apex_agents", "benchmark_parent_name": "APEX Agents", "benchmark_component_key": "management_consulting", "benchmark_component_name": "Management Consulting", "benchmark_leaf_key": "apex_agents", "benchmark_leaf_name": "APEX Agents", "slice_key": "management_consulting", "slice_name": "Management Consulting", "metric_name": "Pass At K", "metric_id": "pass_at_k", "metric_key": "pass_at_k", "metric_source": "metric_config", "display_name": "Management Consulting / Pass At K", "canonical_display_name": "APEX Agents / Management Consulting / Pass At K", "raw_evaluation_name": "Management Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens", "eval_plan", "eval_limits" ], "required_field_count": 4, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Pass At K" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "knowledge": [ { "eval_summary_id": "apex_v1", "benchmark": "APEX v1", "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "benchmark_component_key": "medicine_md", "benchmark_component_name": "Medicine (MD)", "evaluation_name": "APEX v1", "display_name": "APEX v1", "canonical_display_name": "APEX v1", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "benchmark_card": { "benchmark_details": { "name": "APEX-v1", "overview": "APEX-Agents (AI Productivity Index for Agents) measures the ability of AI agents to execute long-horizon, cross-application tasks created by investment banking analysts, management consultants, and corporate lawyers. The benchmark contains 480 tasks and requires agents to navigate realistic work environments with files and tools.", "data_type": "text", "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://arxiv.org/abs/2601.14242", "https://huggingface.co/datasets/Mercor/APEX-v1" ], "benchmark_type": "single" }, "purpose_and_intended_users": { "goal": "To assess whether AI agents can reliably execute highly complex professional services work, bridging the gap between existing agentic evaluations and real-world professional workflows.", "audience": [ "AI researchers", "Developers working on agentic systems" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ], "limitations": "Differences in benchmark scores below 1 percentage point should be interpreted cautiously due to a small error rate in the automated grading system (1.9% false negative rate and 1.3% false positive rate).", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark data was created by industry professionals including investment banking analysts, management consultants, and corporate lawyers. These professionals were organized into teams, assigned specific roles, and tasked with delivering complete projects over 5-10 day periods, producing high-quality customer-ready deliverables from scratch.", "size": "480 tasks", "format": "The specific structure of individual data instances is not described", "annotation": "Tasks were created by professionals using files from within each project environment. A baselining study was conducted where independent experts executed 20% of tasks (96 tasks) to verify task feasibility, rubric fairness, and time estimates." }, "methodology": { "methods": [ "Models are evaluated using agent execution in realistic environments", "Eight trajectories are collected for each agent-task pair, with each trajectory scored as pass or fail" ], "metrics": [ "Pass@1 (task-uniform mean of per-task pass rates)", "Pass@8 (passing at least once in eight attempts)", "Pass^8 (passing consistently on all eight attempts)" ], "calculation": "The overall Pass@1 score is computed as the task-uniform mean of per-task pass rates across all 480 tasks. Confidence intervals are calculated using task-level bootstrapping with 10,000 resamples", "interpretation": "Higher Pass@1 scores indicate better performance", "baseline_results": "Gemini 3 Flash (Thinking=High): 24.0%, GPT-5.2 (Thinking=High): 23.0%, Claude Opus 4.5 (Thinking=High): [score not specified], Gemini 3 Pro (Thinking=High): [score not specified], GPT-OSS-120B (High): 15.2%, Grok 4: 0%", "validation": "Automated evaluation used a judge model with 98.5% accuracy against human-labeled ground truth. A baselining study with experts validated task feasibility and rubric fairness" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Incomplete AI agent evaluation", "description": [ "Evaluating the performance or accuracy or an agent is difficult because of system complexity and open-endedness." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incomplete-ai-agent-evaluation-agentic.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.similar_benchmarks", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-14T14:28:12.501639", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "investment banking", "management consulting", "corporate law", "finance", "legal", "consulting" ], "languages": [ "English" ], "tasks": [ "Text generation", "Question answering", "Reasoning", "Demonstrating advanced knowledge", "Using multiple applications", "Planning over long horizons within realistic project scenarios" ] }, "subtasks_count": 1, "metrics_count": 2, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 2, "has_reproducibility_gap_count": 2, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 2, "total_groups": 2, "multi_source_groups": 0, "first_party_only_groups": 2, "source_type_distribution": { "first_party": 2, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 2, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "apex_v1_score", "legacy_eval_summary_id": "apex_v1_apex_v1", "evaluation_name": "apex-v1", "display_name": "APEX v1 / Score", "canonical_display_name": "APEX v1 / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Overall APEX-v1 mean score (paper snapshot).", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Overall Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.64, "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_v1_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": null, "slice_name": null, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Score", "canonical_display_name": "APEX v1 / Score", "raw_evaluation_name": "apex-v1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [ { "subtask_key": "consulting", "subtask_name": "Consulting", "display_name": "Consulting", "metrics": [ { "metric_summary_id": "apex_v1_consulting_score", "legacy_eval_summary_id": "apex_v1_consulting", "evaluation_name": "Consulting", "display_name": "APEX v1 / Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "lower_is_better": false, "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Management consulting score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Consulting Score" }, "metric_id": "apex_v1.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.64, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.64, "evaluation_id": "apex-v1/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor APEX-v1 Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "apex-v1", "source_type": "hf_dataset", "hf_repo": "Mercor/APEX-v1" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/apex_v1_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "apex_v1", "benchmark_family_name": "APEX v1", "benchmark_parent_key": "apex_v1", "benchmark_parent_name": "APEX v1", "benchmark_component_key": "consulting", "benchmark_component_name": "Consulting", "benchmark_leaf_key": "apex_v1", "benchmark_leaf_name": "APEX v1", "slice_key": "consulting", "slice_name": "Consulting", "metric_name": "Score", "metric_id": "apex_v1.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Consulting / Score", "canonical_display_name": "APEX v1 / Consulting / Score", "raw_evaluation_name": "Consulting", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "other": [ { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "benchmark_component_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_component_name": "artificial_analysis.median_output_tokens_per_second", "evaluation_name": "artificial_analysis.median_output_tokens_per_second", "display_name": "artificial_analysis.median_output_tokens_per_second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median output tokens per second" ], "primary_metric_name": "Median output tokens per second", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second_median_output_tokens_per_second", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_output_tokens_per_second", "evaluation_name": "artificial_analysis.median_output_tokens_per_second", "display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Median output tokens per second", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_key": "median_output_tokens_per_second", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median output generation speed reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_name": "Median output tokens per second", "metric_kind": "throughput", "metric_unit": "tokens_per_second", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 746.589, "additional_details": { "raw_metric_field": "median_output_tokens_per_second", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 181.859, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 181.859, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_component_name": "artificial_analysis.median_output_tokens_per_second", "benchmark_leaf_key": "artificial_analysis_median_output_tokens_per_second", "benchmark_leaf_name": "artificial_analysis.median_output_tokens_per_second", "slice_key": null, "slice_name": null, "metric_name": "Median output tokens per second", "metric_id": "artificial_analysis.median_output_tokens_per_second", "metric_key": "median_output_tokens_per_second", "metric_source": "metric_config", "display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "canonical_display_name": "artificial_analysis.median_output_tokens_per_second / Median output tokens per second", "raw_evaluation_name": "artificial_analysis.median_output_tokens_per_second", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 181.859, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "benchmark_component_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_component_name": "artificial_analysis.median_time_to_first_answer_token", "evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "display_name": "artificial_analysis.median_time_to_first_answer_token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median time to first answer token" ], "primary_metric_name": "Median time to first answer token", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token_median_time_to_first_answer_token", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_answer_token", "evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Median time to first answer token", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_key": "median_time_to_first_answer_token", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median time to first answer token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_name": "Median time to first answer token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_answer_token", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.959, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.959, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_component_name": "artificial_analysis.median_time_to_first_answer_token", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_answer_token", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_answer_token", "slice_key": null, "slice_name": null, "metric_name": "Median time to first answer token", "metric_id": "artificial_analysis.median_time_to_first_answer_token", "metric_key": "median_time_to_first_answer_token", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "canonical_display_name": "artificial_analysis.median_time_to_first_answer_token / Median time to first answer token", "raw_evaluation_name": "artificial_analysis.median_time_to_first_answer_token", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.959, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "benchmark_component_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_component_name": "artificial_analysis.median_time_to_first_token_seconds", "evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "display_name": "artificial_analysis.median_time_to_first_token_seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Median Time To First Token Seconds" ], "primary_metric_name": "Median Time To First Token Seconds", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds_median_time_to_first_token_seconds", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_median_time_to_first_token_seconds", "evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Median Time To First Token Seconds", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_key": "median_time_to_first_token_seconds", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Median time to first token reported by Artificial Analysis.", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_name": "Median time to first token", "metric_kind": "latency", "metric_unit": "seconds", "metric_parameters": { "prompt_length": 1000.0, "parallel_queries": 1.0 }, "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 157.877, "additional_details": { "raw_metric_field": "median_time_to_first_token_seconds", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.959, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.959, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_component_name": "artificial_analysis.median_time_to_first_token_seconds", "benchmark_leaf_key": "artificial_analysis_median_time_to_first_token_seconds", "benchmark_leaf_name": "artificial_analysis.median_time_to_first_token_seconds", "slice_key": null, "slice_name": null, "metric_name": "Median Time To First Token Seconds", "metric_id": "artificial_analysis.median_time_to_first_token_seconds", "metric_key": "median_time_to_first_token_seconds", "metric_source": "metric_config", "display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "canonical_display_name": "artificial_analysis.median_time_to_first_token_seconds / Median Time To First Token Seconds", "raw_evaluation_name": "artificial_analysis.median_time_to_first_token_seconds", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.959, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "benchmark_component_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_component_name": "artificial_analysis.price_1m_blended_3_to_1", "evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "display_name": "artificial_analysis.price_1m_blended_3_to_1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Blended 3 To 1" ], "primary_metric_name": "Price 1m Blended 3 To 1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1_price_1m_blended_3_to_1", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_blended_3_to_1", "evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Blended 3 To 1", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_key": "price_1m_blended_3_to_1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Blended price per 1M tokens using a 3:1 input-to-output ratio.", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_name": "Price per 1M tokens (blended 3:1)", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 262.5, "additional_details": { "raw_metric_field": "price_1m_blended_3_to_1", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 1.125, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 1.125, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_component_name": "artificial_analysis.price_1m_blended_3_to_1", "benchmark_leaf_key": "artificial_analysis_price_1m_blended_3_to_1", "benchmark_leaf_name": "artificial_analysis.price_1m_blended_3_to_1", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Blended 3 To 1", "metric_id": "artificial_analysis.price_1m_blended_3_to_1", "metric_key": "price_1m_blended_3_to_1", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "canonical_display_name": "artificial_analysis.price_1m_blended_3_to_1 / Price 1m Blended 3 To 1", "raw_evaluation_name": "artificial_analysis.price_1m_blended_3_to_1", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 1.125, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "benchmark_component_key": "artificial_analysis_price_1m_input_tokens", "benchmark_component_name": "artificial_analysis.price_1m_input_tokens", "evaluation_name": "artificial_analysis.price_1m_input_tokens", "display_name": "artificial_analysis.price_1m_input_tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Input Tokens" ], "primary_metric_name": "Price 1m Input Tokens", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens_price_1m_input_tokens", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_input_tokens", "evaluation_name": "artificial_analysis.price_1m_input_tokens", "display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Input Tokens", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_key": "price_1m_input_tokens", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Price per 1M input tokens in USD.", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_name": "Price per 1M input tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 150.0, "additional_details": { "raw_metric_field": "price_1m_input_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 0.5, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.5, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_input_tokens", "benchmark_component_name": "artificial_analysis.price_1m_input_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_input_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_input_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Input Tokens", "metric_id": "artificial_analysis.price_1m_input_tokens", "metric_key": "price_1m_input_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "canonical_display_name": "artificial_analysis.price_1m_input_tokens / Price 1m Input Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_input_tokens", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.5, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "benchmark_component_key": "artificial_analysis_price_1m_output_tokens", "benchmark_component_name": "artificial_analysis.price_1m_output_tokens", "evaluation_name": "artificial_analysis.price_1m_output_tokens", "display_name": "artificial_analysis.price_1m_output_tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Price 1m Output Tokens" ], "primary_metric_name": "Price 1m Output Tokens", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens_price_1m_output_tokens", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_price_1m_output_tokens", "evaluation_name": "artificial_analysis.price_1m_output_tokens", "display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "slice_key": null, "slice_name": null, "lower_is_better": true, "metric_name": "Price 1m Output Tokens", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_key": "price_1m_output_tokens", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Price per 1M output tokens in USD.", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_name": "Price per 1M output tokens", "metric_kind": "cost", "metric_unit": "usd_per_1m_tokens", "lower_is_better": true, "score_type": "continuous", "min_score": 0.0, "max_score": 600.0, "additional_details": { "raw_metric_field": "price_1m_output_tokens", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 3.0, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 3.0, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_price_1m_output_tokens", "benchmark_component_name": "artificial_analysis.price_1m_output_tokens", "benchmark_leaf_key": "artificial_analysis_price_1m_output_tokens", "benchmark_leaf_name": "artificial_analysis.price_1m_output_tokens", "slice_key": null, "slice_name": null, "metric_name": "Price 1m Output Tokens", "metric_id": "artificial_analysis.price_1m_output_tokens", "metric_key": "price_1m_output_tokens", "metric_source": "metric_config", "display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "canonical_display_name": "artificial_analysis.price_1m_output_tokens / Price 1m Output Tokens", "raw_evaluation_name": "artificial_analysis.price_1m_output_tokens", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 3.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_intelligence_index", "evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "display_name": "artificial_analysis.artificial_analysis_intelligence_index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Intelligence Index" ], "primary_metric_name": "Artificial Analysis Intelligence Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index_artificial_analysis_intelligence_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_intelligence_index", "evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Intelligence Index", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_key": "artificial_analysis_intelligence_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite intelligence index.", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_name": "Artificial Analysis Intelligence Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.2, "additional_details": { "raw_metric_field": "artificial_analysis_intelligence_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 35.0, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 35.0, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_intelligence_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_intelligence_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_intelligence_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Intelligence Index", "metric_id": "artificial_analysis.artificial_analysis_intelligence_index", "metric_key": "artificial_analysis_intelligence_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "canonical_display_name": "artificial_analysis.artificial_analysis_intelligence_index / Artificial Analysis Intelligence Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_intelligence_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 35.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "benchmark_component_key": "artificial_analysis_gpqa", "benchmark_component_name": "artificial_analysis.gpqa", "evaluation_name": "artificial_analysis.gpqa", "display_name": "artificial_analysis.gpqa", "canonical_display_name": "artificial_analysis.gpqa", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "GPQA" ], "primary_metric_name": "GPQA", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa_gpqa", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_gpqa", "evaluation_name": "artificial_analysis.gpqa", "display_name": "artificial_analysis.gpqa / GPQA", "canonical_display_name": "artificial_analysis.gpqa / GPQA", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "GPQA", "metric_id": "artificial_analysis.gpqa", "metric_key": "gpqa", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on GPQA.", "metric_id": "artificial_analysis.gpqa", "metric_name": "GPQA", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "gpqa", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.812, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.812, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_gpqa", "benchmark_component_name": "artificial_analysis.gpqa", "benchmark_leaf_key": "artificial_analysis_gpqa", "benchmark_leaf_name": "artificial_analysis.gpqa", "slice_key": null, "slice_name": null, "metric_name": "GPQA", "metric_id": "artificial_analysis.gpqa", "metric_key": "gpqa", "metric_source": "metric_config", "display_name": "artificial_analysis.gpqa / GPQA", "canonical_display_name": "artificial_analysis.gpqa / GPQA", "raw_evaluation_name": "artificial_analysis.gpqa", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.812, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_hle", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "benchmark_component_key": "artificial_analysis_hle", "benchmark_component_name": "artificial_analysis.hle", "evaluation_name": "artificial_analysis.hle", "display_name": "artificial_analysis.hle", "canonical_display_name": "artificial_analysis.hle", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Hle" ], "primary_metric_name": "Hle", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_hle_hle", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_hle", "evaluation_name": "artificial_analysis.hle", "display_name": "artificial_analysis.hle / Hle", "canonical_display_name": "artificial_analysis.hle / Hle", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Hle", "metric_id": "artificial_analysis.hle", "metric_key": "hle", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on Humanity's Last Exam.", "metric_id": "artificial_analysis.hle", "metric_name": "Humanity's Last Exam", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "hle", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.141, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.141, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_hle", "benchmark_component_name": "artificial_analysis.hle", "benchmark_leaf_key": "artificial_analysis_hle", "benchmark_leaf_name": "artificial_analysis.hle", "slice_key": null, "slice_name": null, "metric_name": "Hle", "metric_id": "artificial_analysis.hle", "metric_key": "hle", "metric_source": "metric_config", "display_name": "artificial_analysis.hle / Hle", "canonical_display_name": "artificial_analysis.hle / Hle", "raw_evaluation_name": "artificial_analysis.hle", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.141, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_scicode", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "benchmark_component_key": "artificial_analysis_scicode", "benchmark_component_name": "artificial_analysis.scicode", "evaluation_name": "artificial_analysis.scicode", "display_name": "artificial_analysis.scicode", "canonical_display_name": "artificial_analysis.scicode", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "SciCode" ], "primary_metric_name": "SciCode", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_scicode_scicode", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_scicode", "evaluation_name": "artificial_analysis.scicode", "display_name": "artificial_analysis.scicode / SciCode", "canonical_display_name": "artificial_analysis.scicode / SciCode", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "SciCode", "metric_id": "artificial_analysis.scicode", "metric_key": "scicode", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on SciCode.", "metric_id": "artificial_analysis.scicode", "metric_name": "SciCode", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "scicode", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.499, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.499, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_scicode", "benchmark_component_name": "artificial_analysis.scicode", "benchmark_leaf_key": "artificial_analysis_scicode", "benchmark_leaf_name": "artificial_analysis.scicode", "slice_key": null, "slice_name": null, "metric_name": "SciCode", "metric_id": "artificial_analysis.scicode", "metric_key": "scicode", "metric_source": "metric_config", "display_name": "artificial_analysis.scicode / SciCode", "canonical_display_name": "artificial_analysis.scicode / SciCode", "raw_evaluation_name": "artificial_analysis.scicode", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.499, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_coding_index", "evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "display_name": "artificial_analysis.artificial_analysis_coding_index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Coding Index" ], "primary_metric_name": "Artificial Analysis Coding Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index_artificial_analysis_coding_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_coding_index", "evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Coding Index", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_key": "artificial_analysis_coding_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite coding index.", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_name": "Artificial Analysis Coding Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 57.3, "additional_details": { "raw_metric_field": "artificial_analysis_coding_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 37.8, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 37.8, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_coding_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_coding_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_coding_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Coding Index", "metric_id": "artificial_analysis.artificial_analysis_coding_index", "metric_key": "artificial_analysis_coding_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "canonical_display_name": "artificial_analysis.artificial_analysis_coding_index / Artificial Analysis Coding Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_coding_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 37.8, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "benchmark_component_key": "artificial_analysis_ifbench", "benchmark_component_name": "artificial_analysis.ifbench", "evaluation_name": "artificial_analysis.ifbench", "display_name": "artificial_analysis.ifbench", "canonical_display_name": "artificial_analysis.ifbench", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "IFBench" ], "primary_metric_name": "IFBench", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench_ifbench", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_ifbench", "evaluation_name": "artificial_analysis.ifbench", "display_name": "artificial_analysis.ifbench / IFBench", "canonical_display_name": "artificial_analysis.ifbench / IFBench", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "IFBench", "metric_id": "artificial_analysis.ifbench", "metric_key": "ifbench", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on IFBench.", "metric_id": "artificial_analysis.ifbench", "metric_name": "IFBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "ifbench", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.551, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.551, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_ifbench", "benchmark_component_name": "artificial_analysis.ifbench", "benchmark_leaf_key": "artificial_analysis_ifbench", "benchmark_leaf_name": "artificial_analysis.ifbench", "slice_key": null, "slice_name": null, "metric_name": "IFBench", "metric_id": "artificial_analysis.ifbench", "metric_key": "ifbench", "metric_source": "metric_config", "display_name": "artificial_analysis.ifbench / IFBench", "canonical_display_name": "artificial_analysis.ifbench / IFBench", "raw_evaluation_name": "artificial_analysis.ifbench", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.551, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_lcr", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "benchmark_component_key": "artificial_analysis_lcr", "benchmark_component_name": "artificial_analysis.lcr", "evaluation_name": "artificial_analysis.lcr", "display_name": "artificial_analysis.lcr", "canonical_display_name": "artificial_analysis.lcr", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Lcr" ], "primary_metric_name": "Lcr", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_lcr_lcr", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_lcr", "evaluation_name": "artificial_analysis.lcr", "display_name": "artificial_analysis.lcr / Lcr", "canonical_display_name": "artificial_analysis.lcr / Lcr", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Lcr", "metric_id": "artificial_analysis.lcr", "metric_key": "lcr", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on AA-LCR.", "metric_id": "artificial_analysis.lcr", "metric_name": "AA-LCR", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "lcr", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.48, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.48, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_lcr", "benchmark_component_name": "artificial_analysis.lcr", "benchmark_leaf_key": "artificial_analysis_lcr", "benchmark_leaf_name": "artificial_analysis.lcr", "slice_key": null, "slice_name": null, "metric_name": "Lcr", "metric_id": "artificial_analysis.lcr", "metric_key": "lcr", "metric_source": "metric_config", "display_name": "artificial_analysis.lcr / Lcr", "canonical_display_name": "artificial_analysis.lcr / Lcr", "raw_evaluation_name": "artificial_analysis.lcr", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.48, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_tau2", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "benchmark_component_key": "artificial_analysis_tau2", "benchmark_component_name": "artificial_analysis.tau2", "evaluation_name": "artificial_analysis.tau2", "display_name": "artificial_analysis.tau2", "canonical_display_name": "artificial_analysis.tau2", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "tau2" ], "primary_metric_name": "tau2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_tau2_tau2", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_tau2", "evaluation_name": "artificial_analysis.tau2", "display_name": "artificial_analysis.tau2 / tau2", "canonical_display_name": "artificial_analysis.tau2 / tau2", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "tau2", "metric_id": "artificial_analysis.tau2", "metric_key": "tau2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on tau2.", "metric_id": "artificial_analysis.tau2", "metric_name": "tau2", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "tau2", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.433, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.433, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_tau2", "benchmark_component_name": "artificial_analysis.tau2", "benchmark_leaf_key": "artificial_analysis_tau2", "benchmark_leaf_name": "artificial_analysis.tau2", "slice_key": null, "slice_name": null, "metric_name": "tau2", "metric_id": "artificial_analysis.tau2", "metric_key": "tau2", "metric_source": "metric_config", "display_name": "artificial_analysis.tau2 / tau2", "canonical_display_name": "artificial_analysis.tau2 / tau2", "raw_evaluation_name": "artificial_analysis.tau2", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.433, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "benchmark_component_key": "artificial_analysis_terminalbench_hard", "benchmark_component_name": "artificial_analysis.terminalbench_hard", "evaluation_name": "artificial_analysis.terminalbench_hard", "display_name": "artificial_analysis.terminalbench_hard", "canonical_display_name": "artificial_analysis.terminalbench_hard", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Terminalbench Hard" ], "primary_metric_name": "Terminalbench Hard", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard_terminalbench_hard", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_terminalbench_hard", "evaluation_name": "artificial_analysis.terminalbench_hard", "display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "canonical_display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Terminalbench Hard", "metric_id": "artificial_analysis.terminalbench_hard", "metric_key": "terminalbench_hard", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on Terminal-Bench Hard.", "metric_id": "artificial_analysis.terminalbench_hard", "metric_name": "Terminal-Bench Hard", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "terminalbench_hard", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.318, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.318, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_terminalbench_hard", "benchmark_component_name": "artificial_analysis.terminalbench_hard", "benchmark_leaf_key": "artificial_analysis_terminalbench_hard", "benchmark_leaf_name": "artificial_analysis.terminalbench_hard", "slice_key": null, "slice_name": null, "metric_name": "Terminalbench Hard", "metric_id": "artificial_analysis.terminalbench_hard", "metric_key": "terminalbench_hard", "metric_source": "metric_config", "display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "canonical_display_name": "artificial_analysis.terminalbench_hard / Terminalbench Hard", "raw_evaluation_name": "artificial_analysis.terminalbench_hard", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.318, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "benchmark_component_key": "artificial_analysis_mmlu_pro", "benchmark_component_name": "artificial_analysis.mmlu_pro", "evaluation_name": "artificial_analysis.mmlu_pro", "display_name": "artificial_analysis.mmlu_pro", "canonical_display_name": "artificial_analysis.mmlu_pro", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "MMLU-Pro" ], "primary_metric_name": "MMLU-Pro", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro_mmlu_pro", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_mmlu_pro", "evaluation_name": "artificial_analysis.mmlu_pro", "display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "canonical_display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "MMLU-Pro", "metric_id": "artificial_analysis.mmlu_pro", "metric_key": "mmlu_pro", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on MMLU-Pro.", "metric_id": "artificial_analysis.mmlu_pro", "metric_name": "MMLU-Pro", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "mmlu_pro", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.882, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.882, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_mmlu_pro", "benchmark_component_name": "artificial_analysis.mmlu_pro", "benchmark_leaf_key": "artificial_analysis_mmlu_pro", "benchmark_leaf_name": "artificial_analysis.mmlu_pro", "slice_key": null, "slice_name": null, "metric_name": "MMLU-Pro", "metric_id": "artificial_analysis.mmlu_pro", "metric_key": "mmlu_pro", "metric_source": "metric_config", "display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "canonical_display_name": "artificial_analysis.mmlu_pro / MMLU-Pro", "raw_evaluation_name": "artificial_analysis.mmlu_pro", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.882, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "benchmark_component_key": "artificial_analysis_livecodebench", "benchmark_component_name": "artificial_analysis.livecodebench", "evaluation_name": "artificial_analysis.livecodebench", "display_name": "artificial_analysis.livecodebench", "canonical_display_name": "artificial_analysis.livecodebench", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "LiveCodeBench" ], "primary_metric_name": "LiveCodeBench", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench_livecodebench", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_livecodebench", "evaluation_name": "artificial_analysis.livecodebench", "display_name": "artificial_analysis.livecodebench / LiveCodeBench", "canonical_display_name": "artificial_analysis.livecodebench / LiveCodeBench", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "LiveCodeBench", "metric_id": "artificial_analysis.livecodebench", "metric_key": "livecodebench", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on LiveCodeBench.", "metric_id": "artificial_analysis.livecodebench", "metric_name": "LiveCodeBench", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "livecodebench", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.797, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.797, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_livecodebench", "benchmark_component_name": "artificial_analysis.livecodebench", "benchmark_leaf_key": "artificial_analysis_livecodebench", "benchmark_leaf_name": "artificial_analysis.livecodebench", "slice_key": null, "slice_name": null, "metric_name": "LiveCodeBench", "metric_id": "artificial_analysis.livecodebench", "metric_key": "livecodebench", "metric_source": "metric_config", "display_name": "artificial_analysis.livecodebench / LiveCodeBench", "canonical_display_name": "artificial_analysis.livecodebench / LiveCodeBench", "raw_evaluation_name": "artificial_analysis.livecodebench", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.797, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "benchmark_component_key": "artificial_analysis_aime_25", "benchmark_component_name": "artificial_analysis.aime_25", "evaluation_name": "artificial_analysis.aime_25", "display_name": "artificial_analysis.aime_25", "canonical_display_name": "artificial_analysis.aime_25", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Aime 25" ], "primary_metric_name": "Aime 25", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25_aime_25", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_aime_25", "evaluation_name": "artificial_analysis.aime_25", "display_name": "artificial_analysis.aime_25 / Aime 25", "canonical_display_name": "artificial_analysis.aime_25 / Aime 25", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Aime 25", "metric_id": "artificial_analysis.aime_25", "metric_key": "aime_25", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Benchmark score on AIME 2025.", "metric_id": "artificial_analysis.aime_25", "metric_name": "AIME 2025", "metric_kind": "benchmark_score", "metric_unit": "proportion", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_metric_field": "aime_25", "bound_strategy": "fixed" } }, "models_count": 1, "top_score": 0.557, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 0.557, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_aime_25", "benchmark_component_name": "artificial_analysis.aime_25", "benchmark_leaf_key": "artificial_analysis_aime_25", "benchmark_leaf_name": "artificial_analysis.aime_25", "slice_key": null, "slice_name": null, "metric_name": "Aime 25", "metric_id": "artificial_analysis.aime_25", "metric_key": "aime_25", "metric_source": "metric_config", "display_name": "artificial_analysis.aime_25 / Aime 25", "canonical_display_name": "artificial_analysis.aime_25 / Aime 25", "raw_evaluation_name": "artificial_analysis.aime_25", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.557, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index", "benchmark": "Artificial Analysis LLM API", "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "benchmark_component_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_math_index", "evaluation_name": "artificial_analysis.artificial_analysis_math_index", "display_name": "artificial_analysis.artificial_analysis_math_index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Artificial Analysis Math Index" ], "primary_metric_name": "Artificial Analysis Math Index", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index_artificial_analysis_math_index", "legacy_eval_summary_id": "artificial_analysis_llms_artificial_analysis_artificial_analysis_math_index", "evaluation_name": "artificial_analysis.artificial_analysis_math_index", "display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Artificial Analysis Math Index", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_key": "artificial_analysis_math_index", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Artificial Analysis composite math index.", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_name": "Artificial Analysis Math Index", "metric_kind": "index", "metric_unit": "points", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 99.0, "additional_details": { "raw_metric_field": "artificial_analysis_math_index", "bound_strategy": "observed_max_from_snapshot" } }, "models_count": 1, "top_score": 55.7, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash Preview (Non-reasoning)", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-flash", "score": 55.7, "evaluation_id": "artificial-analysis-llms/google/gemini-3-flash/1775918921.622802", "retrieved_timestamp": "1775918921.622802", "source_metadata": { "source_name": "Artificial Analysis LLM API", "source_type": "documentation", "source_organization_name": "Artificial Analysis", "source_organization_url": "https://artificialanalysis.ai/", "evaluator_relationship": "third_party", "additional_details": { "api_endpoint": "https://artificialanalysis.ai/api/v2/data/llms/models", "api_reference_url": "https://artificialanalysis.ai/api-reference", "methodology_url": "https://artificialanalysis.ai/methodology", "attribution_url": "https://artificialanalysis.ai/", "attribution_required": "true", "endpoint_scope": "llms", "prompt_length": "1000", "parallel_queries": "1" } }, "source_data": { "dataset_name": "Artificial Analysis LLM API", "source_type": "url", "url": [ "https://artificialanalysis.ai/api/v2/data/llms/models" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/artificial_analysis_llms_google_gemini_3_flash_1775918921_622802.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "artificial_analysis_llms", "benchmark_family_name": "Artificial Analysis LLM API", "benchmark_parent_key": "artificial_analysis_llms", "benchmark_parent_name": "Artificial Analysis LLM API", "benchmark_component_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_component_name": "artificial_analysis.artificial_analysis_math_index", "benchmark_leaf_key": "artificial_analysis_artificial_analysis_math_index", "benchmark_leaf_name": "artificial_analysis.artificial_analysis_math_index", "slice_key": null, "slice_name": null, "metric_name": "Artificial Analysis Math Index", "metric_id": "artificial_analysis.artificial_analysis_math_index", "metric_key": "artificial_analysis_math_index", "metric_source": "metric_config", "display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "canonical_display_name": "artificial_analysis.artificial_analysis_math_index / Artificial Analysis Math Index", "raw_evaluation_name": "artificial_analysis.artificial_analysis_math_index", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 55.7, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "ace", "benchmark": "ACE", "benchmark_family_key": "ace", "benchmark_family_name": "ACE", "benchmark_parent_key": "ace", "benchmark_parent_name": "ACE", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "ACE", "display_name": "ACE", "canonical_display_name": "ACE", "is_summary_score": false, "category": "other", "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "benchmark_card": { "benchmark_details": { "name": "ace", "overview": "The ACE benchmark measures evaluation criteria across four specific research domains, providing domain-specific evaluation criteria rather than general performance metrics.", "data_type": "tabular, text", "domains": [ "DIY/home improvement", "food/recipes", "shopping/products", "gaming design" ], "languages": [ "Not specified" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://huggingface.co/datasets/Mercor/ACE" ] }, "benchmark_type": "single", "purpose_and_intended_users": { "goal": "Not specified", "audience": [ "Not specified" ], "tasks": [ "Not specified" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "Not specified", "size": "Fewer than 1,000 examples", "format": "CSV", "annotation": "Not specified" }, "methodology": { "methods": [ "Not specified" ], "metrics": [ "Overall Score", "Gaming Score", "DIY Score", "Food Score", "Shopping Score" ], "calculation": "No explicit calculation method provided in structured metadata", "interpretation": "Higher scores indicate better performance for all metrics.", "baseline_results": "Evaluation results from Every Eval Ever: GPT 5 achieved 0.5965 (average_across_subjects), GPT 5.2 scored 0.5810, o3 Pro scored 0.5510, GPT 5.1 scored 0.5428, and o3 scored 0.5213. Across 12 evaluated models, scores ranged from 0.332 to 0.5965 with a mean of 0.4643 and standard deviation of 0.0947.", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Creative Commons Attribution 4.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Incorrect risk testing", "description": [ "A metric selected to measure or track a risk is incorrectly selected, incompletely measuring the risk, or measuring the wrong risk for the given context." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/incorrect-risk-testing.html" }, { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "benchmark_details.languages", "benchmark_details.similar_benchmarks", "purpose_and_intended_users.goal", "purpose_and_intended_users.audience", "purpose_and_intended_users.tasks", "purpose_and_intended_users.limitations", "purpose_and_intended_users.out_of_scope_uses", "data.source", "data.annotation", "methodology.methods", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-04-13T01:07:34.053021", "llm": "deepseek-ai/DeepSeek-V3.1" } }, "tags": { "domains": [ "DIY/home improvement", "food/recipes", "shopping/products", "gaming design" ], "languages": [ "Not specified" ], "tasks": [ "Not specified" ] }, "subtasks_count": 1, "metrics_count": 1, "metric_names": [ "Score" ], "primary_metric_name": "Score", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 1, "source_type_distribution": { "first_party": 1, "third_party": 0, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [], "subtasks": [ { "subtask_key": "gaming", "subtask_name": "Gaming", "display_name": "Gaming", "metrics": [ { "metric_summary_id": "ace_gaming_score", "legacy_eval_summary_id": "ace_gaming", "evaluation_name": "Gaming", "display_name": "ACE / Gaming / Score", "canonical_display_name": "ACE / Gaming / Score", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "slice_key": "gaming", "slice_name": "Gaming", "lower_is_better": false, "metric_name": "Score", "metric_id": "ace.score", "metric_key": "score", "metric_source": "metric_config", "metric_config": { "evaluation_description": "Gaming domain score.", "lower_is_better": false, "score_type": "continuous", "min_score": 0, "max_score": 1, "additional_details": { "raw_evaluation_name": "Gaming Score" }, "metric_id": "ace.score", "metric_name": "Score", "metric_kind": "score", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.415, "model_results": [ { "model_id": "google/gemini-3-flash", "model_route_id": "google__gemini-3-flash", "model_name": "Gemini 3 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/Gemini 3 Flash", "score": 0.415, "evaluation_id": "ace/google_gemini-3-flash/1773260200", "retrieved_timestamp": "1773260200", "source_metadata": { "source_name": "Mercor ACE Leaderboard", "source_type": "evaluation_run", "source_organization_name": "Mercor", "source_organization_url": "https://www.mercor.com", "evaluator_relationship": "first_party" }, "source_data": { "dataset_name": "ace", "source_type": "hf_dataset", "hf_repo": "Mercor/ACE" }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-flash/ace_google_gemini_3_flash_1773260200.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "ace", "benchmark_family_name": "ACE", "benchmark_parent_key": "ace", "benchmark_parent_name": "ACE", "benchmark_component_key": "gaming", "benchmark_component_name": "Gaming", "benchmark_leaf_key": "ace", "benchmark_leaf_name": "ACE", "slice_key": "gaming", "slice_name": "Gaming", "metric_name": "Score", "metric_id": "ace.score", "metric_key": "score", "metric_source": "metric_config", "display_name": "Gaming / Score", "canonical_display_name": "ACE / Gaming / Score", "raw_evaluation_name": "Gaming", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "first_party", "is_multi_source": false, "first_party_only": true, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "metrics_count": 1, "metric_names": [ "Score" ] } ], "models_count": 1, "top_score": null, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "total_evaluations": 8, "last_updated": "2026-04-11T14:48:41.622802Z", "categories_covered": [ "agentic", "knowledge", "other" ], "variants": [ { "variant_key": "default", "variant_label": "Default", "evaluation_count": 8, "raw_model_ids": [ "google/Gemini 3 Flash", "google/gemini-3-flash" ], "last_updated": "2026-04-11T14:48:41.622802Z" } ], "reproducibility_summary": { "results_total": 33, "has_reproducibility_gap_count": 33, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 33, "total_groups": 29, "multi_source_groups": 0, "first_party_only_groups": 9, "source_type_distribution": { "first_party": 10, "third_party": 23, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 29, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 } }