{ "eval_summary_id": "helm_lite_narrativeqa", "benchmark": "NarrativeQA", "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA", "canonical_display_name": "NarrativeQA", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "NarrativeQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_lite", "overview": "helm_lite is a composite benchmark suite designed to provide a comprehensive evaluation of AI models across multiple domains and capabilities. It aggregates performance metrics from nine distinct sub-benchmarks: GSM8K, LegalBench, MATH, MMLU, MedQA, NarrativeQA, NaturalQuestions (closed-book), OpenbookQA, and WMT 2014. The suite measures overall model proficiency through aggregated scoring across these diverse tasks, covering areas such as mathematical reasoning, legal analysis, multilingual understanding, medical knowledge, narrative comprehension, and question answering.", "benchmark_type": "composite", "contains": [ "GSM8K", "LegalBench", "MATH", "MMLU", "MedQA", "NarrativeQA", "NaturalQuestions (closed-book)", "OpenbookQA", "WMT 2014" ], "data_type": "composite", "domains": [ "mathematical reasoning", "legal analysis", "multilingual understanding", "medical knowledge", "narrative comprehension", "question answering" ], "languages": [ "English", "multiple" ], "similar_benchmarks": [ "Not specified" ], "resources": [ "https://crfm.stanford.edu/helm/lite/latest/", "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To provide a comprehensive evaluation of AI models across multiple reasoning domains and knowledge areas through a standardized composite benchmark suite.", "audience": [ "AI researchers", "Model developers" ], "tasks": [ "Mathematical reasoning", "Legal reasoning", "Medical question answering", "General knowledge evaluation", "Reading comprehension", "Machine translation" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The HELM-Lite benchmark is a composite suite that aggregates multiple specialized benchmarks to measure overall AI performance across diverse domains including reasoning, knowledge, and language understanding. It integrates GSM8K (math), LegalBench (legal reasoning), MATH (advanced math), MMLU (multidisciplinary knowledge), MedQA (medical knowledge), NarrativeQA (story understanding), NaturalQuestions (closed-book factoid QA), OpenbookQA (commonsense reasoning), and WMT 2014 (machine translation).", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The suite evaluates AI models across multiple reasoning domains using a composite of established benchmarks", "Each sub-benchmark measures specific capabilities including mathematical reasoning, legal analysis, medical knowledge, reading comprehension, and machine translation" ], "metrics": [ "Exact Match (EM) for GSM8K, LegalBench, MMLU, MedQA, and OpenbookQA", "F1 score for NarrativeQA and NaturalQuestions (closed-book)", "Equivalent (CoT) for MATH", "BLEU-4 for WMT 2014" ], "calculation": "Scores are calculated independently for each sub-benchmark using their respective metrics and aggregated at the suite level. All metrics are continuous with higher scores indicating better performance.", "interpretation": "Higher scores across all metrics indicate better overall performance. The suite provides comprehensive coverage across mathematical, legal, medical, commonsense reasoning, reading comprehension, and translation capabilities.", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "mathematical reasoning", "legal analysis", "multilingual understanding", "medical knowledge", "narrative comprehension", "question answering" ], "languages": [ "English", "multiple" ], "tasks": [ "Mathematical reasoning", "Legal reasoning", "Medical question answering", "General knowledge evaluation", "Reading comprehension", "Machine translation" ] }, "subtasks": [], "metrics": [ { "metric_summary_id": "helm_lite_narrativeqa_f1", "legacy_eval_summary_id": "helm_lite_narrativeqa", "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA / F1", "canonical_display_name": "NarrativeQA / F1", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on NarrativeQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "model_results": [ { "model_id": "openai/gpt-4o", "model_route_id": "openai__gpt-4o", "model_name": "GPT-4o 2024-05-13", "developer": "openai", "variant_key": "2024-05-13", "raw_model_id": "openai/gpt-4o-2024-05-13", "score": 0.804, "evaluation_id": "helm_lite/openai_gpt-4o-2024-05-13/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o/helm_lite_openai_gpt_4o_2024_05_13_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-70b", "model_route_id": "meta__llama-3-70b", "model_name": "Llama 3 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3-70b", "score": 0.798, "evaluation_id": "helm_lite/meta_llama-3-70b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-70b/helm_lite_meta_llama_3_70b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-v3", "model_route_id": "deepseek-ai__deepseek-v3", "model_name": "DeepSeek v3", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-v3", "score": 0.796, "evaluation_id": "helm_lite/deepseek-ai_deepseek-v3/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-v3/helm_lite_deepseek_ai_deepseek_v3_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4o", "model_route_id": "openai__gpt-4o", "model_name": "GPT-4o 2024-08-06", "developer": "openai", "variant_key": "2024-08-06", "raw_model_id": "openai/gpt-4o-2024-08-06", "score": 0.795, "evaluation_id": "helm_lite/openai_gpt-4o-2024-08-06/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o/helm_lite_openai_gpt_4o_2024_08_06_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-3-70b-instruct-turbo", "model_route_id": "meta__llama-3-3-70b-instruct-turbo", "model_name": "Llama 3.3 Instruct Turbo 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.3-70b-instruct-turbo", "score": 0.791, "evaluation_id": "helm_lite/meta_llama-3.3-70b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-3-70b-instruct-turbo/helm_lite_meta_llama_3_3_70b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "amazon/nova-pro-v1-0", "model_route_id": "amazon__nova-pro-v1-0", "model_name": "Amazon Nova Pro", "developer": "amazon", "variant_key": "default", "raw_model_id": "amazon/nova-pro-v1:0", "score": 0.791, "evaluation_id": "helm_lite/amazon_nova-pro-v1:0/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/amazon__nova-pro-v1-0/helm_lite_amazon_nova_pro_v1_0_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemma-2-27b-it", "model_route_id": "google__gemma-2-27b-it", "model_name": "Gemma 2 Instruct 27B", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemma-2-27b-it", "score": 0.79, "evaluation_id": "helm_lite/google_gemma-2-27b-it/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemma-2-27b-it/helm_lite_google_gemma_2_27b_it_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-0-flash-exp", "model_route_id": "google__gemini-2-0-flash-exp", "model_name": "Gemini 2.0 Flash Experimental", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.0-flash-exp", "score": 0.783, "evaluation_id": "helm_lite/google_gemini-2.0-flash-exp/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-exp/helm_lite_google_gemini_2_0_flash_exp_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-pro-001", "model_route_id": "google__gemini-1-5-pro-001", "model_name": "Gemini 1.5 Pro 001", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-pro-001", "score": 0.783, "evaluation_id": "helm_lite/google_gemini-1.5-pro-001/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-pro-001/helm_lite_google_gemini_1_5_pro_001_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-flash-001", "model_route_id": "google__gemini-1-5-flash-001", "model_name": "Gemini 1.5 Flash 001", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-flash-001", "score": 0.783, "evaluation_id": "helm_lite/google_gemini-1.5-flash-001/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-flash-001/helm_lite_google_gemini_1_5_flash_001_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "01-ai/yi-34b", "model_route_id": "01-ai__yi-34b", "model_name": "Yi 34B", "developer": "01-ai", "variant_key": "default", "raw_model_id": "01-ai/yi-34b", "score": 0.782, "evaluation_id": "helm_lite/01-ai_yi-34b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/01-ai__yi-34b/helm_lite_01_ai_yi_34b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mixtral-8x22b", "model_route_id": "mistralai__mixtral-8x22b", "model_name": "Mixtral 8x22B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mixtral-8x22b", "score": 0.779, "evaluation_id": "helm_lite/mistralai_mixtral-8x22b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x22b/helm_lite_mistralai_mixtral_8x22b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-large-2407", "model_route_id": "mistralai__mistral-large-2407", "model_name": "Mistral Large 2 2407", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-large-2407", "score": 0.779, "evaluation_id": "helm_lite/mistralai_mistral-large-2407/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-large-2407/helm_lite_mistralai_mistral_large_2407_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-2-90b-vision-instruct-turbo", "model_route_id": "meta__llama-3-2-90b-vision-instruct-turbo", "model_name": "Llama 3.2 Vision Instruct Turbo 90B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.2-90b-vision-instruct-turbo", "score": 0.777, "evaluation_id": "helm_lite/meta_llama-3.2-90b-vision-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-2-90b-vision-instruct-turbo/helm_lite_meta_llama_3_2_90b_vision_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-x-004", "model_route_id": "writer__palmyra-x-004", "model_name": "Palmyra-X-004", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-x-004", "score": 0.773, "evaluation_id": "helm_lite/writer_palmyra-x-004/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-004/helm_lite_writer_palmyra_x_004_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-70b-instruct-turbo", "model_route_id": "meta__llama-3-1-70b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-70b-instruct-turbo", "score": 0.772, "evaluation_id": "helm_lite/meta_llama-3.1-70b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-70b-instruct-turbo/helm_lite_meta_llama_3_1_70b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-5-sonnet", "model_route_id": "anthropic__claude-3-5-sonnet", "model_name": "Claude 3.5 Sonnet 20241022", "developer": "anthropic", "variant_key": "20241022", "raw_model_id": "anthropic/claude-3-5-sonnet-20241022", "score": 0.77, "evaluation_id": "helm_lite/anthropic_claude-3-5-sonnet-20241022/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-sonnet/helm_lite_anthropic_claude_3_5_sonnet_20241022_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4o-mini", "model_route_id": "openai__gpt-4o-mini", "model_name": "GPT-4o mini 2024-07-18", "developer": "openai", "variant_key": "2024-07-18", "raw_model_id": "openai/gpt-4o-mini-2024-07-18", "score": 0.768, "evaluation_id": "helm_lite/openai_gpt-4o-mini-2024-07-18/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o-mini/helm_lite_openai_gpt_4o_mini_2024_07_18_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-0613", "model_route_id": "openai__gpt-4-0613", "model_name": "GPT-4 0613", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-4-0613", "score": 0.768, "evaluation_id": "helm_lite/openai_gpt-4-0613/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-0613/helm_lite_openai_gpt_4_0613_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemma-2-9b-it", "model_route_id": "google__gemma-2-9b-it", "model_name": "Gemma 2 Instruct 9B", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemma-2-9b-it", "score": 0.768, "evaluation_id": "helm_lite/google_gemma-2-9b-it/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemma-2-9b-it/helm_lite_google_gemma_2_9b_it_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "amazon/nova-lite-v1-0", "model_route_id": "amazon__nova-lite-v1-0", "model_name": "Amazon Nova Lite", "developer": "amazon", "variant_key": "default", "raw_model_id": "amazon/nova-lite-v1:0", "score": 0.768, "evaluation_id": "helm_lite/amazon_nova-lite-v1:0/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/amazon__nova-lite-v1-0/helm_lite_amazon_nova_lite_v1_0_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mixtral-8x7b-32kseqlen", "model_route_id": "mistralai__mixtral-8x7b-32kseqlen", "model_name": "Mixtral 8x7B 32K seqlen", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mixtral-8x7b-32kseqlen", "score": 0.767, "evaluation_id": "helm_lite/mistralai_mixtral-8x7b-32kseqlen/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x7b-32kseqlen/helm_lite_mistralai_mixtral_8x7b_32kseqlen_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-2-70b", "model_route_id": "meta__llama-2-70b", "model_name": "Llama 2 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-2-70b", "score": 0.763, "evaluation_id": "helm_lite/meta_llama-2-70b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-2-70b/helm_lite_meta_llama_2_70b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-5-haiku", "model_route_id": "anthropic__claude-3-5-haiku", "model_name": "Claude 3.5 Haiku 20241022", "developer": "anthropic", "variant_key": "20241022", "raw_model_id": "anthropic/claude-3-5-haiku-20241022", "score": 0.763, "evaluation_id": "helm_lite/anthropic_claude-3-5-haiku-20241022/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-haiku/helm_lite_anthropic_claude_3_5_haiku_20241022_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-turbo", "model_route_id": "openai__gpt-4-turbo", "model_name": "GPT-4 Turbo 2024-04-09", "developer": "openai", "variant_key": "2024-04-09", "raw_model_id": "openai/gpt-4-turbo-2024-04-09", "score": 0.761, "evaluation_id": "helm_lite/openai_gpt-4-turbo-2024-04-09/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-turbo/helm_lite_openai_gpt_4_turbo_2024_04_09_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-2-11b-vision-instruct-turbo", "model_route_id": "meta__llama-3-2-11b-vision-instruct-turbo", "model_name": "Llama 3.2 Vision Instruct Turbo 11B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.2-11b-vision-instruct-turbo", "score": 0.756, "evaluation_id": "helm_lite/meta_llama-3.2-11b-vision-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-2-11b-vision-instruct-turbo/helm_lite_meta_llama_3_2_11b_vision_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-8b-instruct-turbo", "model_route_id": "meta__llama-3-1-8b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 8B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-8b-instruct-turbo", "score": 0.756, "evaluation_id": "helm_lite/meta_llama-3.1-8b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-8b-instruct-turbo/helm_lite_meta_llama_3_1_8b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-pro-002", "model_route_id": "google__gemini-1-5-pro-002", "model_name": "Gemini 1.5 Pro 002", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-pro-002", "score": 0.756, "evaluation_id": "helm_lite/google_gemini-1.5-pro-002/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-pro-002/helm_lite_google_gemini_1_5_pro_002_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-65b", "model_route_id": "meta__llama-65b", "model_name": "LLaMA 65B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-65b", "score": 0.755, "evaluation_id": "helm_lite/meta_llama-65b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-65b/helm_lite_meta_llama_65b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "microsoft/phi-3-small-8k-instruct", "model_route_id": "microsoft__phi-3-small-8k-instruct", "model_name": "Phi-3 7B", "developer": "microsoft", "variant_key": "default", "raw_model_id": "microsoft/phi-3-small-8k-instruct", "score": 0.754, "evaluation_id": "helm_lite/microsoft_phi-3-small-8k-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/microsoft__phi-3-small-8k-instruct/helm_lite_microsoft_phi_3_small_8k_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-8b", "model_route_id": "meta__llama-3-8b", "model_name": "Llama 3 8B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3-8b", "score": 0.754, "evaluation_id": "helm_lite/meta_llama-3-8b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-8b/helm_lite_meta_llama_3_8b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "upstage/solar-pro-241126", "model_route_id": "upstage__solar-pro-241126", "model_name": "Solar Pro", "developer": "upstage", "variant_key": "default", "raw_model_id": "upstage/solar-pro-241126", "score": 0.753, "evaluation_id": "helm_lite/upstage_solar-pro-241126/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/upstage__solar-pro-241126/helm_lite_upstage_solar_pro_241126_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-x-v2", "model_route_id": "writer__palmyra-x-v2", "model_name": "Palmyra X V2 33B", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-x-v2", "score": 0.752, "evaluation_id": "helm_lite/writer_palmyra-x-v2/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-v2/helm_lite_writer_palmyra_x_v2_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemma-7b", "model_route_id": "google__gemma-7b", "model_name": "Gemma 7B", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemma-7b", "score": 0.752, "evaluation_id": "helm_lite/google_gemma-7b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemma-7b/helm_lite_google_gemma_7b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-0-pro-002", "model_route_id": "google__gemini-1-0-pro-002", "model_name": "Gemini 1.0 Pro 002", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.0-pro-002", "score": 0.751, "evaluation_id": "helm_lite/google_gemini-1.0-pro-002/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-0-pro-002/helm_lite_google_gemini_1_0_pro_002_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-405b-instruct-turbo", "model_route_id": "meta__llama-3-1-405b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 405B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-405b-instruct-turbo", "score": 0.749, "evaluation_id": "helm_lite/meta_llama-3.1-405b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-405b-instruct-turbo/helm_lite_meta_llama_3_1_405b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command", "model_route_id": "cohere__command", "model_name": "Command", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command", "score": 0.749, "evaluation_id": "helm_lite/cohere_command/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command/helm_lite_cohere_command_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-flash-002", "model_route_id": "google__gemini-1-5-flash-002", "model_name": "Gemini 1.5 Flash 002", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-flash-002", "score": 0.746, "evaluation_id": "helm_lite/google_gemini-1.5-flash-002/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-flash-002/helm_lite_google_gemini_1_5_flash_002_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-5-sonnet", "model_route_id": "anthropic__claude-3-5-sonnet", "model_name": "Claude 3.5 Sonnet 20240620", "developer": "anthropic", "variant_key": "20240620", "raw_model_id": "anthropic/claude-3-5-sonnet-20240620", "score": 0.746, "evaluation_id": "helm_lite/anthropic_claude-3-5-sonnet-20240620/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-sonnet/helm_lite_anthropic_claude_3_5_sonnet_20240620_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ai21/jamba-1-5-mini", "model_route_id": "ai21__jamba-1-5-mini", "model_name": "Jamba 1.5 Mini", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/jamba-1.5-mini", "score": 0.746, "evaluation_id": "helm_lite/ai21_jamba-1.5-mini/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__jamba-1-5-mini/helm_lite_ai21_jamba_1_5_mini_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-5-72b-instruct-turbo", "model_route_id": "qwen__qwen2-5-72b-instruct-turbo", "model_name": "Qwen2.5 Instruct Turbo 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2.5-72b-instruct-turbo", "score": 0.745, "evaluation_id": "helm_lite/qwen_qwen2.5-72b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-72b-instruct-turbo/helm_lite_qwen_qwen2_5_72b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "amazon/nova-micro-v1-0", "model_route_id": "amazon__nova-micro-v1-0", "model_name": "Amazon Nova Micro", "developer": "amazon", "variant_key": "default", "raw_model_id": "amazon/nova-micro-v1:0", "score": 0.744, "evaluation_id": "helm_lite/amazon_nova-micro-v1:0/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/amazon__nova-micro-v1-0/helm_lite_amazon_nova_micro_v1_0_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ai21/j2-grande", "model_route_id": "ai21__j2-grande", "model_name": "Jurassic-2 Grande 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/j2-grande", "score": 0.744, "evaluation_id": "helm_lite/ai21_j2-grande/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j2-grande/helm_lite_ai21_j2_grande_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "alephalpha/luminous-supreme", "model_route_id": "alephalpha__luminous-supreme", "model_name": "Luminous Supreme 70B", "developer": "AlephAlpha", "variant_key": "default", "raw_model_id": "AlephAlpha/luminous-supreme", "score": 0.743, "evaluation_id": "helm_lite/AlephAlpha_luminous-supreme/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/alephalpha__luminous-supreme/helm_lite_alephalpha_luminous_supreme_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-5-7b-instruct-turbo", "model_route_id": "qwen__qwen2-5-7b-instruct-turbo", "model_name": "Qwen2.5 Instruct Turbo 7B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2.5-7b-instruct-turbo", "score": 0.742, "evaluation_id": "helm_lite/qwen_qwen2.5-7b-instruct-turbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-7b-instruct-turbo/helm_lite_qwen_qwen2_5_7b_instruct_turbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command-r", "model_route_id": "cohere__command-r", "model_name": "Command R", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command-r", "score": 0.742, "evaluation_id": "helm_lite/cohere_command-r/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r/helm_lite_cohere_command_r_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-2-13b", "model_route_id": "meta__llama-2-13b", "model_name": "Llama 2 13B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-2-13b", "score": 0.741, "evaluation_id": "helm_lite/meta_llama-2-13b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-2-13b/helm_lite_meta_llama_2_13b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command-r-plus", "model_route_id": "cohere__command-r-plus", "model_name": "Command R Plus", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command-r-plus", "score": 0.735, "evaluation_id": "helm_lite/cohere_command-r-plus/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r-plus/helm_lite_cohere_command_r_plus_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/text-davinci-003", "model_route_id": "openai__text-davinci-003", "model_name": "GPT-3.5 text-davinci-003", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/text-davinci-003", "score": 0.731, "evaluation_id": "helm_lite/openai_text-davinci-003/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__text-davinci-003/helm_lite_openai_text_davinci_003_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/open-mistral-nemo-2407", "model_route_id": "mistralai__open-mistral-nemo-2407", "model_name": "Mistral NeMo 2402", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/open-mistral-nemo-2407", "score": 0.731, "evaluation_id": "helm_lite/mistralai_open-mistral-nemo-2407/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__open-mistral-nemo-2407/helm_lite_mistralai_open_mistral_nemo_2407_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ai21/j2-jumbo", "model_route_id": "ai21__j2-jumbo", "model_name": "Jurassic-2 Jumbo 178B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/j2-jumbo", "score": 0.728, "evaluation_id": "helm_lite/ai21_j2-jumbo/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j2-jumbo/helm_lite_ai21_j2_jumbo_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-72b-instruct", "model_route_id": "qwen__qwen2-72b-instruct", "model_name": "Qwen2 Instruct 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2-72b-instruct", "score": 0.727, "evaluation_id": "helm_lite/qwen_qwen2-72b-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-72b-instruct/helm_lite_qwen_qwen2_72b_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-1106-preview", "model_route_id": "openai__gpt-4-1106-preview", "model_name": "GPT-4 Turbo 1106 preview", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-4-1106-preview", "score": 0.727, "evaluation_id": "helm_lite/openai_gpt-4-1106-preview/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1106-preview/helm_lite_openai_gpt_4_1106_preview_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "microsoft/phi-3-medium-4k-instruct", "model_route_id": "microsoft__phi-3-medium-4k-instruct", "model_name": "Phi-3 14B", "developer": "microsoft", "variant_key": "default", "raw_model_id": "microsoft/phi-3-medium-4k-instruct", "score": 0.724, "evaluation_id": "helm_lite/microsoft_phi-3-medium-4k-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/microsoft__phi-3-medium-4k-instruct/helm_lite_microsoft_phi_3_medium_4k_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-v1-3", "model_route_id": "anthropic__claude-v1-3", "model_name": "Claude v1.3", "developer": "anthropic", "variant_key": "default", "raw_model_id": "anthropic/claude-v1.3", "score": 0.723, "evaluation_id": "helm_lite/anthropic_claude-v1.3/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-v1-3/helm_lite_anthropic_claude_v1_3_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-110b-chat", "model_route_id": "qwen__qwen1-5-110b-chat", "model_name": "Qwen1.5 Chat 110B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-110b-chat", "score": 0.721, "evaluation_id": "helm_lite/qwen_qwen1.5-110b-chat/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-110b-chat/helm_lite_qwen_qwen1_5_110b_chat_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/text-davinci-002", "model_route_id": "openai__text-davinci-002", "model_name": "GPT-3.5 text-davinci-002", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/text-davinci-002", "score": 0.719, "evaluation_id": "helm_lite/openai_text-davinci-002/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__text-davinci-002/helm_lite_openai_text_davinci_002_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/text-bison-001", "model_route_id": "google__text-bison-001", "model_name": "PaLM-2 Bison", "developer": "google", "variant_key": "default", "raw_model_id": "google/text-bison@001", "score": 0.718, "evaluation_id": "helm_lite/google_text-bison@001/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__text-bison-001/helm_lite_google_text_bison_001_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-2-0", "model_route_id": "anthropic__claude-2-0", "model_name": "Claude 2.0", "developer": "anthropic", "variant_key": "default", "raw_model_id": "anthropic/claude-2.0", "score": 0.718, "evaluation_id": "helm_lite/anthropic_claude-2.0/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-2-0/helm_lite_anthropic_claude_2_0_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-7b-v0-1", "model_route_id": "mistralai__mistral-7b-v0-1", "model_name": "Mistral v0.1 7B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-7b-v0.1", "score": 0.716, "evaluation_id": "helm_lite/mistralai_mistral-7b-v0.1/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-v0-1/helm_lite_mistralai_mistral_7b_v0_1_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-7b-instruct-v0-3", "model_route_id": "mistralai__mistral-7b-instruct-v0-3", "model_name": "Mistral Instruct v0.3 7B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-7b-instruct-v0.3", "score": 0.716, "evaluation_id": "helm_lite/mistralai_mistral-7b-instruct-v0.3/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-3/helm_lite_mistralai_mistral_7b_instruct_v0_3_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-14b", "model_route_id": "qwen__qwen1-5-14b", "model_name": "Qwen1.5 14B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-14b", "score": 0.711, "evaluation_id": "helm_lite/qwen_qwen1.5-14b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-14b/helm_lite_qwen_qwen1_5_14b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-x-v3", "model_route_id": "writer__palmyra-x-v3", "model_name": "Palmyra X V3 72B", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-x-v3", "score": 0.706, "evaluation_id": "helm_lite/writer_palmyra-x-v3/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-v3/helm_lite_writer_palmyra_x_v3_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "microsoft/phi-2", "model_route_id": "microsoft__phi-2", "model_name": "Phi-2", "developer": "microsoft", "variant_key": "default", "raw_model_id": "microsoft/phi-2", "score": 0.703, "evaluation_id": "helm_lite/microsoft_phi-2/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/microsoft__phi-2/helm_lite_microsoft_phi_2_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "01-ai/yi-6b", "model_route_id": "01-ai__yi-6b", "model_name": "Yi 6B", "developer": "01-ai", "variant_key": "default", "raw_model_id": "01-ai/yi-6b", "score": 0.702, "evaluation_id": "helm_lite/01-ai_yi-6b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/01-ai__yi-6b/helm_lite_01_ai_yi_6b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-2-7b", "model_route_id": "meta__llama-2-7b", "model_name": "Llama 2 7B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-2-7b", "score": 0.686, "evaluation_id": "helm_lite/meta_llama-2-7b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-2-7b/helm_lite_meta_llama_2_7b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "alephalpha/luminous-extended", "model_route_id": "alephalpha__luminous-extended", "model_name": "Luminous Extended 30B", "developer": "AlephAlpha", "variant_key": "default", "raw_model_id": "AlephAlpha/luminous-extended", "score": 0.684, "evaluation_id": "helm_lite/AlephAlpha_luminous-extended/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/alephalpha__luminous-extended/helm_lite_alephalpha_luminous_extended_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-2-1", "model_route_id": "anthropic__claude-2-1", "model_name": "Claude 2.1", "developer": "anthropic", "variant_key": "default", "raw_model_id": "anthropic/claude-2.1", "score": 0.677, "evaluation_id": "helm_lite/anthropic_claude-2.1/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-2-1/helm_lite_anthropic_claude_2_1_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "tiiuae/falcon-40b", "model_route_id": "tiiuae__falcon-40b", "model_name": "Falcon 40B", "developer": "tiiuae", "variant_key": "default", "raw_model_id": "tiiuae/falcon-40b", "score": 0.671, "evaluation_id": "helm_lite/tiiuae_falcon-40b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/tiiuae__falcon-40b/helm_lite_tiiuae_falcon_40b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ai21/jamba-1-5-large", "model_route_id": "ai21__jamba-1-5-large", "model_name": "Jamba 1.5 Large", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/jamba-1.5-large", "score": 0.664, "evaluation_id": "helm_lite/ai21_jamba-1.5-large/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__jamba-1-5-large/helm_lite_ai21_jamba_1_5_large_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ai21/jamba-instruct", "model_route_id": "ai21__jamba-instruct", "model_name": "Jamba Instruct", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/jamba-instruct", "score": 0.658, "evaluation_id": "helm_lite/ai21_jamba-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__jamba-instruct/helm_lite_ai21_jamba_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-3-5-turbo-0613", "model_route_id": "openai__gpt-3-5-turbo-0613", "model_name": "GPT-3.5 Turbo 0613", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-3.5-turbo-0613", "score": 0.655, "evaluation_id": "helm_lite/openai_gpt-3.5-turbo-0613/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0613/helm_lite_openai_gpt_3_5_turbo_0613_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "snowflake/snowflake-arctic-instruct", "model_route_id": "snowflake__snowflake-arctic-instruct", "model_name": "Arctic Instruct", "developer": "snowflake", "variant_key": "default", "raw_model_id": "snowflake/snowflake-arctic-instruct", "score": 0.654, "evaluation_id": "helm_lite/snowflake_snowflake-arctic-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/snowflake__snowflake-arctic-instruct/helm_lite_snowflake_snowflake_arctic_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "alephalpha/luminous-base", "model_route_id": "alephalpha__luminous-base", "model_name": "Luminous Base 13B", "developer": "AlephAlpha", "variant_key": "default", "raw_model_id": "AlephAlpha/luminous-base", "score": 0.633, "evaluation_id": "helm_lite/AlephAlpha_luminous-base/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/alephalpha__luminous-base/helm_lite_alephalpha_luminous_base_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command-light", "model_route_id": "cohere__command-light", "model_name": "Command Light", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command-light", "score": 0.629, "evaluation_id": "helm_lite/cohere_command-light/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-light/helm_lite_cohere_command_light_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "tiiuae/falcon-7b", "model_route_id": "tiiuae__falcon-7b", "model_name": "Falcon 7B", "developer": "tiiuae", "variant_key": "default", "raw_model_id": "tiiuae/falcon-7b", "score": 0.621, "evaluation_id": "helm_lite/tiiuae_falcon-7b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/tiiuae__falcon-7b/helm_lite_tiiuae_falcon_7b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-instant-1-2", "model_route_id": "anthropic__claude-instant-1-2", "model_name": "Claude Instant 1.2", "developer": "anthropic", "variant_key": "default", "raw_model_id": "anthropic/claude-instant-1.2", "score": 0.616, "evaluation_id": "helm_lite/anthropic_claude-instant-1.2/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-instant-1-2/helm_lite_anthropic_claude_instant_1_2_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-72b", "model_route_id": "qwen__qwen1-5-72b", "model_name": "Qwen1.5 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-72b", "score": 0.601, "evaluation_id": "helm_lite/qwen_qwen1.5-72b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-72b/helm_lite_qwen_qwen1_5_72b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "allenai/olmo-7b", "model_route_id": "allenai__olmo-7b", "model_name": "OLMo 7B", "developer": "allenai", "variant_key": "default", "raw_model_id": "allenai/olmo-7b", "score": 0.597, "evaluation_id": "helm_lite/allenai_olmo-7b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-7b/helm_lite_allenai_olmo_7b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-32b", "model_route_id": "qwen__qwen1-5-32b", "model_name": "Qwen1.5 32B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-32b", "score": 0.589, "evaluation_id": "helm_lite/qwen_qwen1.5-32b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-32b/helm_lite_qwen_qwen1_5_32b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/text-unicorn-001", "model_route_id": "google__text-unicorn-001", "model_name": "PaLM-2 Unicorn", "developer": "google", "variant_key": "default", "raw_model_id": "google/text-unicorn@001", "score": 0.583, "evaluation_id": "helm_lite/google_text-unicorn@001/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__text-unicorn-001/helm_lite_google_text_unicorn_001_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-llm-67b-chat", "model_route_id": "deepseek-ai__deepseek-llm-67b-chat", "model_name": "DeepSeek LLM Chat 67B", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-llm-67b-chat", "score": 0.581, "evaluation_id": "helm_lite/deepseek-ai_deepseek-llm-67b-chat/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-llm-67b-chat/helm_lite_deepseek_ai_deepseek_llm_67b_chat_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-small-2402", "model_route_id": "mistralai__mistral-small-2402", "model_name": "Mistral Small 2402", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-small-2402", "score": 0.519, "evaluation_id": "helm_lite/mistralai_mistral-small-2402/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-small-2402/helm_lite_mistralai_mistral_small_2402_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "databricks/dbrx-instruct", "model_route_id": "databricks__dbrx-instruct", "model_name": "DBRX Instruct", "developer": "databricks", "variant_key": "default", "raw_model_id": "databricks/dbrx-instruct", "score": 0.488, "evaluation_id": "helm_lite/databricks_dbrx-instruct/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/databricks__dbrx-instruct/helm_lite_databricks_dbrx_instruct_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-large-2402", "model_route_id": "mistralai__mistral-large-2402", "model_name": "Mistral Large 2402", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-large-2402", "score": 0.454, "evaluation_id": "helm_lite/mistralai_mistral-large-2402/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-large-2402/helm_lite_mistralai_mistral_large_2402_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-medium-2312", "model_route_id": "mistralai__mistral-medium-2312", "model_name": "Mistral Medium 2312", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-medium-2312", "score": 0.449, "evaluation_id": "helm_lite/mistralai_mistral-medium-2312/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-medium-2312/helm_lite_mistralai_mistral_medium_2312_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-7b", "model_route_id": "qwen__qwen1-5-7b", "model_name": "Qwen1.5 7B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-7b", "score": 0.448, "evaluation_id": "helm_lite/qwen_qwen1.5-7b/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-7b/helm_lite_qwen_qwen1_5_7b_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "01-ai/yi-large-preview", "model_route_id": "01-ai__yi-large-preview", "model_name": "Yi Large Preview", "developer": "01-ai", "variant_key": "default", "raw_model_id": "01-ai/yi-large-preview", "score": 0.373, "evaluation_id": "helm_lite/01-ai_yi-large-preview/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/01-ai__yi-large-preview/helm_lite_01_ai_yi_large_preview_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-opus", "model_route_id": "anthropic__claude-3-opus", "model_name": "Claude 3 Opus 20240229", "developer": "anthropic", "variant_key": "20240229", "raw_model_id": "anthropic/claude-3-opus-20240229", "score": 0.351, "evaluation_id": "helm_lite/anthropic_claude-3-opus-20240229/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-opus/helm_lite_anthropic_claude_3_opus_20240229_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-haiku", "model_route_id": "anthropic__claude-3-haiku", "model_name": "Claude 3 Haiku 20240307", "developer": "anthropic", "variant_key": "20240307", "raw_model_id": "anthropic/claude-3-haiku-20240307", "score": 0.244, "evaluation_id": "helm_lite/anthropic_claude-3-haiku-20240307/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-haiku/helm_lite_anthropic_claude_3_haiku_20240307_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-sonnet", "model_route_id": "anthropic__claude-3-sonnet", "model_name": "Claude 3 Sonnet 20240229", "developer": "anthropic", "variant_key": "20240229", "raw_model_id": "anthropic/claude-3-sonnet-20240229", "score": 0.111, "evaluation_id": "helm_lite/anthropic_claude-3-sonnet-20240229/1774096306.427425", "retrieved_timestamp": "1774096306.427425", "source_metadata": { "source_name": "helm_lite", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_lite", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-sonnet/helm_lite_anthropic_claude_3_sonnet_20240229_1774096306_427425.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_lite", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_lite", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ], "models_count": 91, "top_score": 0.804 } ], "subtasks_count": 0, "metrics_count": 1, "models_count": 89, "metric_names": [ "F1" ], "primary_metric_name": "F1", "top_score": 0.804, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 }, "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 91, "has_reproducibility_gap_count": 91, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 91, "total_groups": 89, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 91, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 89, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 } }