{ "benchmark_card": { "benchmark_details": { "name": "Omni-MATH", "overview": "Omni-MATH is a benchmark designed to measure the mathematical reasoning capabilities of large language models at the Olympiad competition level. It comprises 4,428 challenging problems, categorized into over 33 sub-domains and more than 10 distinct difficulty levels. It was created because existing mathematical benchmarks are nearing saturation and are inadequate for assessing models on truly difficult tasks.", "data_type": "text", "domains": [ "math", "olympiads" ], "languages": [ "English" ], "similar_benchmarks": [ "GSM8K", "MATH" ], "resources": [ "https://arxiv.org/abs/2410.07985", "https://huggingface.co/datasets/KbsdJames/Omni-MATH", "https://storage.googleapis.com/crfm-helm-public/capabilities/benchmark_output/releases/v1.15.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To provide a challenging benchmark for assessing and advancing the mathematical reasoning capabilities of large language models at the Olympiad and competition level, as existing benchmarks have become less challenging. It enables a nuanced analysis of performance across various mathematical disciplines and complexity levels.", "audience": [ "Researchers evaluating large language models" ], "tasks": [ "Solving Olympiad-level mathematical problems", "Solving competition-level mathematical problems", "Rule-based evaluation on a filtered subset of problems (Omni-MATH-Rule)" ], "limitations": "Some problems are not suitable for reliable rule-based evaluation, leading to a filtered 'testable' set and an 'untestable' set.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The data is sourced from contest pages, the AoPS Wiki, and the AoPS Forum. Initial raw counts from these sources were filtered to create the final dataset.", "size": "4,428 problems. A subset of 2,821 problems (Omni-MATH-Rule) is suitable for rule-based evaluation, and a separate subset of 100 samples was used for meta-evaluation.", "format": "JSON", "annotation": "Problems underwent rigorous human annotation by a team including PhD and Master's students. Each problem was reviewed by two annotators for cross-validation, achieving high agreement rates." }, "methodology": { "methods": [ "Models are evaluated by generating solutions to the mathematical problems.", "Evaluation is conducted using both model-based judgment (e.g., GPT-4o or the open-source Omni-Judge) and rule-based evaluation for a specific subset of problems (Omni-MATH-Rule)." ], "metrics": [ "Accuracy (Acc)" ], "calculation": "The overall score is the accuracy percentage, calculated as the number of correct solutions divided by the total number of problems.", "interpretation": "Higher accuracy indicates stronger performance. The benchmark is considered challenging, as even advanced models achieve only moderate accuracy.", "baseline_results": "Paper baselines (accuracy on full Omni-MATH / Omni-MATH-Rule subset): o1-mini (60.54% / 62.2%), o1-preview (52.55% / 51.7%), qwen2.5-MATH-72b-Instruct (36.20% / 35.7%), qwen2.5-MATH-7b-Instruct (33.22% / 32.3%), GPT-4o (30.49% / 29.2%), NuminaMATH-72b-cot (28.45% / 27.1%), DeepseekMATH-7b-RL (16.12% / 14.9%). EEE results: OLMo 2 32B Instruct March 2025 scored 0.161 (16.1%).", "validation": "For the Omni-MATH-Rule subset, two PhD students annotated problems to determine suitability for rule-based evaluation, achieving 98% cross-validation accuracy. A meta-evaluation compared model-generated judgments against human gold-standard annotations on a 100-sample subset to assess reliability." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation.\u00a0" ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T13:34:44.331592", "llm": "deepseek-ai/DeepSeek-V3.2" } } }