{ "model_info": { "name": "J1-Grande v1 17B", "id": "ai21/j1-grande-v1-17b", "developer": "ai21", "inference_platform": "unknown", "normalized_id": "ai21/J1-Grande-v1-17B", "family_id": "ai21/j1-grande-v1-17b", "family_slug": "j1-grande-v1-17b", "family_name": "J1-Grande v1 17B", "variant_key": "default", "variant_label": "Default", "model_route_id": "ai21__j1-grande-v1-17b", "model_version": null }, "model_group_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_family_name": "J1-Grande v1 17B", "raw_model_ids": [ "ai21/J1-Grande-v1-17B" ], "evaluations_by_category": { "general": [ { "schema_version": "0.2.2", "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "benchmark": "helm_classic", "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "eval_library": { "name": "helm", "version": "unknown" }, "model_info": { "name": "J1-Grande v1 17B", "id": "ai21/J1-Grande-v1-17B", "developer": "ai21", "inference_platform": "unknown", "normalized_id": "ai21/J1-Grande-v1-17B", "family_id": "ai21/j1-grande-v1-17b", "family_slug": "j1-grande-v1-17b", "family_name": "J1-Grande v1 17B", "variant_key": "default", "variant_label": "Default", "model_route_id": "ai21__j1-grande-v1-17b" }, "generation_config": { "additional_details": {} }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results_meta": null, "detailed_evaluation_results": null, "passthrough_top_level_fields": null, "evaluation_results": [ { "evaluation_name": "helm_classic", "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "How many models this model outperform on average (over columns).", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_evaluation_name": "Mean win rate" }, "metric_id": "win_rate", "metric_name": "Win Rate", "metric_kind": "win_rate", "metric_unit": "proportion" }, "score_details": { "score": 0.433, "details": { "description": "", "tab": "Accuracy", "Mean win rate - Calibration": "{\"description\": \"\", \"tab\": \"Calibration\", \"score\": \"0.6221919576066971\"}", "Mean win rate - Robustness": "{\"description\": \"\", \"tab\": \"Robustness\", \"score\": \"0.4225080073800875\"}", "Mean win rate - Fairness": "{\"description\": \"\", \"tab\": \"Fairness\", \"score\": \"0.4539316449216338\"}", "Mean win rate - Efficiency": "{\"description\": \"\", \"tab\": \"Efficiency\", \"score\": \"0.31716008771929827\"}", "Mean win rate - General information": "{\"description\": \"\", \"tab\": \"General information\", \"score\": \"\"}", "Mean win rate - Bias": "{\"description\": \"\", \"tab\": \"Bias\", \"score\": \"0.5580147362700336\"}", "Mean win rate - Toxicity": "{\"description\": \"\", \"tab\": \"Toxicity\", \"score\": \"0.6300489633822968\"}", "Mean win rate - Summarization metrics": "{\"description\": \"\", \"tab\": \"Summarization metrics\", \"score\": \"0.6689640768588138\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#helm_classic#win_rate", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "Helm classic", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "Helm classic", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "slice_key": null, "slice_name": null, "metric_name": "Win Rate", "metric_id": "win_rate", "metric_key": "win_rate", "metric_source": "metric_config", "display_name": "Win Rate", "canonical_display_name": "Helm classic / Win Rate", "raw_evaluation_name": "helm_classic", "is_summary_score": false } }, { "evaluation_name": "MMLU", "source_data": { "dataset_name": "MMLU", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on MMLU", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.27, "details": { "description": "min=0.2, mean=0.27, max=0.35, sum=4.047 (15)", "tab": "Accuracy", "MMLU - ECE (10-bin)": "{\"description\": \"min=0.063, mean=0.114, max=0.154, sum=1.708 (15)\", \"tab\": \"Calibration\", \"score\": \"0.11389257817699022\"}", "MMLU - EM (Robustness)": "{\"description\": \"min=0.15, mean=0.225, max=0.27, sum=3.377 (15)\", \"tab\": \"Robustness\", \"score\": \"0.22511111111111112\"}", "MMLU - EM (Fairness)": "{\"description\": \"min=0.158, mean=0.232, max=0.29, sum=3.474 (15)\", \"tab\": \"Fairness\", \"score\": \"0.23159064327485382\"}", "MMLU - Denoised inference time (s)": "{\"description\": \"min=0.381, mean=0.411, max=0.466, sum=6.166 (15)\", \"tab\": \"Efficiency\", \"score\": \"0.41104061293859656\"}", "MMLU - # eval": "{\"description\": \"min=100, mean=102.8, max=114, sum=1542 (15)\", \"tab\": \"General information\", \"score\": \"102.8\"}", "MMLU - # train": "{\"description\": \"min=5, mean=5, max=5, sum=75 (15)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "MMLU - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (15)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "MMLU - # prompt tokens": "{\"description\": \"min=308.59, mean=396.74, max=552.719, sum=5951.098 (15)\", \"tab\": \"General information\", \"score\": \"396.73985964912276\"}", "MMLU - # output tokens": "{\"description\": \"min=1, mean=1, max=1, sum=15 (15)\", \"tab\": \"General information\", \"score\": \"1.0\"}", "MMLU - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=45 (15)\", \"tab\": \"General information\", \"score\": \"3.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#mmlu#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MMLU", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MMLU", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "MMLU / Exact Match", "raw_evaluation_name": "MMLU", "is_summary_score": false } }, { "evaluation_name": "BoolQ", "source_data": { "dataset_name": "BoolQ", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on BoolQ", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.722, "details": { "description": "min=0.712, mean=0.722, max=0.733, sum=2.165 (3)", "tab": "Accuracy", "BoolQ - ECE (10-bin)": "{\"description\": \"min=0.139, mean=0.154, max=0.169, sum=0.462 (3)\", \"tab\": \"Calibration\", \"score\": \"0.15409092997354776\"}", "BoolQ - EM (Robustness)": "{\"description\": \"min=0.632, mean=0.643, max=0.658, sum=1.929 (3)\", \"tab\": \"Robustness\", \"score\": \"0.6429999999999999\"}", "BoolQ - EM (Fairness)": "{\"description\": \"min=0.656, mean=0.678, max=0.695, sum=2.035 (3)\", \"tab\": \"Fairness\", \"score\": \"0.6783333333333333\"}", "BoolQ - Denoised inference time (s)": "{\"description\": \"min=0.47, mean=0.535, max=0.624, sum=1.606 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.5352501416015627\"}", "BoolQ - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "BoolQ - # train": "{\"description\": \"min=5, mean=5, max=5, sum=15 (3)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "BoolQ - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "BoolQ - # prompt tokens": "{\"description\": \"min=506.985, mean=694.652, max=952.985, sum=2083.955 (3)\", \"tab\": \"General information\", \"score\": \"694.6516666666666\"}", "BoolQ - # output tokens": "{\"description\": \"min=2, mean=2, max=2, sum=6 (3)\", \"tab\": \"General information\", \"score\": \"2.0\"}", "BoolQ - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "BoolQ - Stereotypes (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "BoolQ - Stereotypes (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "BoolQ - Representation (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "BoolQ - Representation (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "BoolQ - Toxic fraction": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"Toxicity\", \"score\": \"0.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#boolq#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "BoolQ", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "BoolQ", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "BoolQ / Exact Match", "raw_evaluation_name": "BoolQ", "is_summary_score": false } }, { "evaluation_name": "NarrativeQA", "source_data": { "dataset_name": "NarrativeQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "F1 on NarrativeQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "score_details": { "score": 0.672, "details": { "description": "min=0.664, mean=0.672, max=0.68, sum=2.016 (3)", "tab": "Accuracy", "NarrativeQA - ECE (10-bin)": "{\"description\": \"min=0.039, mean=0.047, max=0.062, sum=0.141 (3)\", \"tab\": \"Calibration\", \"score\": \"0.04705310707412085\"}", "NarrativeQA - F1 (Robustness)": "{\"description\": \"min=0.409, mean=0.477, max=0.522, sum=1.432 (3)\", \"tab\": \"Robustness\", \"score\": \"0.47749086119263257\"}", "NarrativeQA - F1 (Fairness)": "{\"description\": \"min=0.526, mean=0.547, max=0.563, sum=1.641 (3)\", \"tab\": \"Fairness\", \"score\": \"0.5469545337986748\"}", "NarrativeQA - Denoised inference time (s)": "{\"description\": \"min=0.892, mean=0.923, max=0.955, sum=2.769 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.9228662338615026\"}", "NarrativeQA - # eval": "{\"description\": \"min=355, mean=355, max=355, sum=1065 (3)\", \"tab\": \"General information\", \"score\": \"355.0\"}", "NarrativeQA - # train": "{\"description\": \"min=2.166, mean=2.639, max=3.225, sum=7.918 (3)\", \"tab\": \"General information\", \"score\": \"2.63943661971831\"}", "NarrativeQA - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "NarrativeQA - # prompt tokens": "{\"description\": \"min=1598.614, mean=1692.218, max=1777.299, sum=5076.654 (3)\", \"tab\": \"General information\", \"score\": \"1692.2178403755868\"}", "NarrativeQA - # output tokens": "{\"description\": \"min=4.324, mean=4.528, max=4.701, sum=13.583 (3)\", \"tab\": \"General information\", \"score\": \"4.527699530516432\"}", "NarrativeQA - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "NarrativeQA - Stereotypes (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "NarrativeQA - Stereotypes (gender)": "{\"description\": \"min=0.5, mean=0.5, max=0.5, sum=1 (2)\", \"tab\": \"Bias\", \"score\": \"0.5\"}", "NarrativeQA - Representation (race)": "{\"description\": \"min=0.667, mean=0.667, max=0.667, sum=1.333 (2)\", \"tab\": \"Bias\", \"score\": \"0.6666666666666667\"}", "NarrativeQA - Representation (gender)": "{\"description\": \"min=0.15, mean=0.164, max=0.18, sum=0.491 (3)\", \"tab\": \"Bias\", \"score\": \"0.1636261091893518\"}", "NarrativeQA - Toxic fraction": "{\"description\": \"min=0.008, mean=0.014, max=0.017, sum=0.042 (3)\", \"tab\": \"Toxicity\", \"score\": \"0.014084507042253521\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#narrativeqa#f1", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false } }, { "evaluation_name": "NaturalQuestions (open-book)", "source_data": { "dataset_name": "NaturalQuestions (open-book)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "F1 on NaturalQuestions (open-book)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "score_details": { "score": 0.578, "details": { "description": "min=0.561, mean=0.578, max=0.59, sum=1.734 (3)", "tab": "Accuracy", "NaturalQuestions (closed-book) - ECE (10-bin)": "{\"description\": \"min=0.027, mean=0.029, max=0.03, sum=0.087 (3)\", \"tab\": \"Calibration\", \"score\": \"0.028955351873343083\"}", "NaturalQuestions (open-book) - ECE (10-bin)": "{\"description\": \"min=0.073, mean=0.081, max=0.097, sum=0.243 (3)\", \"tab\": \"Calibration\", \"score\": \"0.08114120238748938\"}", "NaturalQuestions (closed-book) - F1 (Robustness)": "{\"description\": \"min=0.164, mean=0.17, max=0.175, sum=0.511 (3)\", \"tab\": \"Robustness\", \"score\": \"0.17025794044565556\"}", "NaturalQuestions (open-book) - F1 (Robustness)": "{\"description\": \"min=0.449, mean=0.478, max=0.494, sum=1.433 (3)\", \"tab\": \"Robustness\", \"score\": \"0.4776074011626843\"}", "NaturalQuestions (closed-book) - F1 (Fairness)": "{\"description\": \"min=0.185, mean=0.187, max=0.189, sum=0.562 (3)\", \"tab\": \"Fairness\", \"score\": \"0.1872477522460834\"}", "NaturalQuestions (open-book) - F1 (Fairness)": "{\"description\": \"min=0.501, mean=0.521, max=0.534, sum=1.563 (3)\", \"tab\": \"Fairness\", \"score\": \"0.5209919156580172\"}", "NaturalQuestions (closed-book) - Denoised inference time (s)": "{\"description\": \"min=0.437, mean=0.466, max=0.494, sum=1.399 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.46640491796874967\"}", "NaturalQuestions (open-book) - Denoised inference time (s)": "{\"description\": \"min=0.774, mean=0.873, max=0.927, sum=2.618 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.8728225097656246\"}", "NaturalQuestions (closed-book) - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "NaturalQuestions (closed-book) - # train": "{\"description\": \"min=5, mean=5, max=5, sum=15 (3)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "NaturalQuestions (closed-book) - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "NaturalQuestions (closed-book) - # prompt tokens": "{\"description\": \"min=94.377, mean=99.377, max=102.377, sum=298.131 (3)\", \"tab\": \"General information\", \"score\": \"99.377\"}", "NaturalQuestions (closed-book) - # output tokens": "{\"description\": \"min=4.791, mean=5.971, max=7.18, sum=17.913 (3)\", \"tab\": \"General information\", \"score\": \"5.971\"}", "NaturalQuestions (closed-book) - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "NaturalQuestions (open-book) - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "NaturalQuestions (open-book) - # train": "{\"description\": \"min=4.568, mean=4.666, max=4.734, sum=13.999 (3)\", \"tab\": \"General information\", \"score\": \"4.666333333333333\"}", "NaturalQuestions (open-book) - truncated": "{\"description\": \"min=0.038, mean=0.038, max=0.038, sum=0.114 (3)\", \"tab\": \"General information\", \"score\": \"0.038\"}", "NaturalQuestions (open-book) - # prompt tokens": "{\"description\": \"min=1136.933, mean=1418.457, max=1595.508, sum=4255.37 (3)\", \"tab\": \"General information\", \"score\": \"1418.4566666666667\"}", "NaturalQuestions (open-book) - # output tokens": "{\"description\": \"min=6.302, mean=6.538, max=6.976, sum=19.615 (3)\", \"tab\": \"General information\", \"score\": \"6.538333333333333\"}", "NaturalQuestions (open-book) - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "NaturalQuestions (closed-book) - Stereotypes (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "NaturalQuestions (closed-book) - Stereotypes (gender)": "{\"description\": \"min=0.5, mean=0.5, max=0.5, sum=1.5 (3)\", \"tab\": \"Bias\", \"score\": \"0.5\"}", "NaturalQuestions (closed-book) - Representation (race)": "{\"description\": \"min=0.473, mean=0.521, max=0.556, sum=1.564 (3)\", \"tab\": \"Bias\", \"score\": \"0.5214747518446415\"}", "NaturalQuestions (closed-book) - Representation (gender)": "{\"description\": \"min=0, mean=0.033, max=0.1, sum=0.1 (3)\", \"tab\": \"Bias\", \"score\": \"0.033333333333333326\"}", "NaturalQuestions (open-book) - Stereotypes (race)": "{\"description\": \"min=0.667, mean=0.667, max=0.667, sum=0.667 (1)\", \"tab\": \"Bias\", \"score\": \"0.6666666666666667\"}", "NaturalQuestions (open-book) - Stereotypes (gender)": "{\"description\": \"min=0.346, mean=0.346, max=0.346, sum=1.038 (3)\", \"tab\": \"Bias\", \"score\": \"0.3461538461538461\"}", "NaturalQuestions (open-book) - Representation (race)": "{\"description\": \"min=0.45, mean=0.488, max=0.521, sum=1.463 (3)\", \"tab\": \"Bias\", \"score\": \"0.48764942579375564\"}", "NaturalQuestions (open-book) - Representation (gender)": "{\"description\": \"min=0.111, mean=0.113, max=0.118, sum=0.34 (3)\", \"tab\": \"Bias\", \"score\": \"0.11339991677070331\"}", "NaturalQuestions (closed-book) - Toxic fraction": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"Toxicity\", \"score\": \"0.0\"}", "NaturalQuestions (open-book) - Toxic fraction": "{\"description\": \"min=0, mean=0.001, max=0.001, sum=0.002 (3)\", \"tab\": \"Toxicity\", \"score\": \"0.0006666666666666666\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#naturalquestions_open_book#f1", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NaturalQuestions (open-book)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NaturalQuestions (open-book)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NaturalQuestions (open-book) / F1", "raw_evaluation_name": "NaturalQuestions (open-book)", "is_summary_score": false } }, { "evaluation_name": "QuAC", "source_data": { "dataset_name": "QuAC", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "F1 on QuAC", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "score_details": { "score": 0.362, "details": { "description": "min=0.355, mean=0.362, max=0.372, sum=1.087 (3)", "tab": "Accuracy", "QuAC - ECE (10-bin)": "{\"description\": \"min=0.019, mean=0.036, max=0.06, sum=0.107 (3)\", \"tab\": \"Calibration\", \"score\": \"0.03571925908384949\"}", "QuAC - F1 (Robustness)": "{\"description\": \"min=0.215, mean=0.219, max=0.227, sum=0.658 (3)\", \"tab\": \"Robustness\", \"score\": \"0.21921244416502939\"}", "QuAC - F1 (Fairness)": "{\"description\": \"min=0.266, mean=0.274, max=0.282, sum=0.821 (3)\", \"tab\": \"Fairness\", \"score\": \"0.27362985580399246\"}", "QuAC - Denoised inference time (s)": "{\"description\": \"min=1.302, mean=1.413, max=1.478, sum=4.24 (3)\", \"tab\": \"Efficiency\", \"score\": \"1.4134776341145843\"}", "QuAC - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "QuAC - # train": "{\"description\": \"min=1.788, mean=1.829, max=1.88, sum=5.486 (3)\", \"tab\": \"General information\", \"score\": \"1.8286666666666667\"}", "QuAC - truncated": "{\"description\": \"min=0.001, mean=0.001, max=0.001, sum=0.003 (3)\", \"tab\": \"General information\", \"score\": \"0.001\"}", "QuAC - # prompt tokens": "{\"description\": \"min=1645.856, mean=1698.711, max=1730.814, sum=5096.134 (3)\", \"tab\": \"General information\", \"score\": \"1698.7113333333334\"}", "QuAC - # output tokens": "{\"description\": \"min=22.154, mean=27.786, max=31.692, sum=83.357 (3)\", \"tab\": \"General information\", \"score\": \"27.785666666666668\"}", "QuAC - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "QuAC - Stereotypes (race)": "{\"description\": \"min=0.58, mean=0.6, max=0.639, sum=1.799 (3)\", \"tab\": \"Bias\", \"score\": \"0.5996635891593876\"}", "QuAC - Stereotypes (gender)": "{\"description\": \"min=0.415, mean=0.428, max=0.44, sum=1.283 (3)\", \"tab\": \"Bias\", \"score\": \"0.42780085419627883\"}", "QuAC - Representation (race)": "{\"description\": \"min=0.298, mean=0.34, max=0.378, sum=1.019 (3)\", \"tab\": \"Bias\", \"score\": \"0.3397817992618246\"}", "QuAC - Representation (gender)": "{\"description\": \"min=0.237, mean=0.242, max=0.25, sum=0.727 (3)\", \"tab\": \"Bias\", \"score\": \"0.24231770708576347\"}", "QuAC - Toxic fraction": "{\"description\": \"min=0.004, mean=0.004, max=0.004, sum=0.012 (3)\", \"tab\": \"Toxicity\", \"score\": \"0.004\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#quac#f1", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "QuAC", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "QuAC", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "QuAC / F1", "raw_evaluation_name": "QuAC", "is_summary_score": false } }, { "evaluation_name": "HellaSwag", "source_data": { "dataset_name": "HellaSwag", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on HellaSwag", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.739, "details": { "description": "min=0.739, mean=0.739, max=0.739, sum=0.739 (1)", "tab": "Accuracy", "HellaSwag - ECE (10-bin)": "{\"description\": \"min=0.213, mean=0.213, max=0.213, sum=0.213 (1)\", \"tab\": \"Calibration\", \"score\": \"0.21338082493857388\"}", "HellaSwag - EM (Robustness)": "{\"description\": \"min=0.695, mean=0.695, max=0.695, sum=0.695 (1)\", \"tab\": \"Robustness\", \"score\": \"0.695\"}", "HellaSwag - EM (Fairness)": "{\"description\": \"min=0.58, mean=0.58, max=0.58, sum=0.58 (1)\", \"tab\": \"Fairness\", \"score\": \"0.58\"}", "HellaSwag - Denoised inference time (s)": "{\"description\": \"min=0.33, mean=0.33, max=0.33, sum=0.33 (1)\", \"tab\": \"Efficiency\", \"score\": \"0.3304377109375\"}", "HellaSwag - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=1000 (1)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "HellaSwag - # train": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "HellaSwag - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "HellaSwag - # prompt tokens": "{\"description\": \"min=62.466, mean=62.466, max=62.466, sum=62.466 (1)\", \"tab\": \"General information\", \"score\": \"62.466\"}", "HellaSwag - # output tokens": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "HellaSwag - # trials": "{\"description\": \"min=1, mean=1, max=1, sum=1 (1)\", \"tab\": \"General information\", \"score\": \"1.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#hellaswag#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "HellaSwag", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "HellaSwag", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "HellaSwag / Exact Match", "raw_evaluation_name": "HellaSwag", "is_summary_score": false } }, { "evaluation_name": "OpenbookQA", "source_data": { "dataset_name": "OpenbookQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on OpenbookQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.52, "details": { "description": "min=0.52, mean=0.52, max=0.52, sum=0.52 (1)", "tab": "Accuracy", "OpenbookQA - ECE (10-bin)": "{\"description\": \"min=0.258, mean=0.258, max=0.258, sum=0.258 (1)\", \"tab\": \"Calibration\", \"score\": \"0.25849314658751343\"}", "OpenbookQA - EM (Robustness)": "{\"description\": \"min=0.424, mean=0.424, max=0.424, sum=0.424 (1)\", \"tab\": \"Robustness\", \"score\": \"0.424\"}", "OpenbookQA - EM (Fairness)": "{\"description\": \"min=0.472, mean=0.472, max=0.472, sum=0.472 (1)\", \"tab\": \"Fairness\", \"score\": \"0.472\"}", "OpenbookQA - Denoised inference time (s)": "{\"description\": \"min=0.281, mean=0.281, max=0.281, sum=0.281 (1)\", \"tab\": \"Efficiency\", \"score\": \"0.280719578125\"}", "OpenbookQA - # eval": "{\"description\": \"min=500, mean=500, max=500, sum=500 (1)\", \"tab\": \"General information\", \"score\": \"500.0\"}", "OpenbookQA - # train": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "OpenbookQA - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "OpenbookQA - # prompt tokens": "{\"description\": \"min=4.348, mean=4.348, max=4.348, sum=4.348 (1)\", \"tab\": \"General information\", \"score\": \"4.348\"}", "OpenbookQA - # output tokens": "{\"description\": \"min=0, mean=0, max=0, sum=0 (1)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "OpenbookQA - # trials": "{\"description\": \"min=1, mean=1, max=1, sum=1 (1)\", \"tab\": \"General information\", \"score\": \"1.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#openbookqa#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "OpenbookQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "OpenbookQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "OpenbookQA / Exact Match", "raw_evaluation_name": "OpenbookQA", "is_summary_score": false } }, { "evaluation_name": "TruthfulQA", "source_data": { "dataset_name": "TruthfulQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on TruthfulQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.193, "details": { "description": "min=0.171, mean=0.193, max=0.217, sum=0.58 (3)", "tab": "Accuracy", "TruthfulQA - ECE (10-bin)": "{\"description\": \"min=0.064, mean=0.091, max=0.109, sum=0.273 (3)\", \"tab\": \"Calibration\", \"score\": \"0.09083831911084679\"}", "TruthfulQA - EM (Robustness)": "{\"description\": \"min=0.116, mean=0.142, max=0.159, sum=0.425 (3)\", \"tab\": \"Robustness\", \"score\": \"0.1416921508664628\"}", "TruthfulQA - EM (Fairness)": "{\"description\": \"min=0.138, mean=0.163, max=0.182, sum=0.489 (3)\", \"tab\": \"Fairness\", \"score\": \"0.16309887869520898\"}", "TruthfulQA - Denoised inference time (s)": "{\"description\": \"min=0.384, mean=0.396, max=0.403, sum=1.189 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.39626294915902127\"}", "TruthfulQA - # eval": "{\"description\": \"min=654, mean=654, max=654, sum=1962 (3)\", \"tab\": \"General information\", \"score\": \"654.0\"}", "TruthfulQA - # train": "{\"description\": \"min=5, mean=5, max=5, sum=15 (3)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "TruthfulQA - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "TruthfulQA - # prompt tokens": "{\"description\": \"min=317.682, mean=355.015, max=375.682, sum=1065.046 (3)\", \"tab\": \"General information\", \"score\": \"355.0152905198777\"}", "TruthfulQA - # output tokens": "{\"description\": \"min=1, mean=1, max=1, sum=3 (3)\", \"tab\": \"General information\", \"score\": \"1.0\"}", "TruthfulQA - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#truthfulqa#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "TruthfulQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "TruthfulQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "TruthfulQA / Exact Match", "raw_evaluation_name": "TruthfulQA", "is_summary_score": false } }, { "evaluation_name": "MS MARCO (TREC)", "source_data": { "dataset_name": "MS MARCO (TREC)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "NDCG@10 on MS MARCO (TREC)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "ndcg", "metric_name": "NDCG@10", "metric_kind": "ndcg", "metric_unit": "proportion", "metric_parameters": { "k": 10 } }, "score_details": { "score": 0.341, "details": { "description": "min=0.31, mean=0.341, max=0.389, sum=1.022 (3)", "tab": "Accuracy", "MS MARCO (regular) - RR@10 (Robustness)": "{\"description\": \"min=0.105, mean=0.121, max=0.133, sum=0.362 (3)\", \"tab\": \"Robustness\", \"score\": \"0.12069748677248683\"}", "MS MARCO (TREC) - NDCG@10 (Robustness)": "{\"description\": \"min=0.27, mean=0.297, max=0.328, sum=0.89 (3)\", \"tab\": \"Robustness\", \"score\": \"0.29680328755123014\"}", "MS MARCO (regular) - RR@10 (Fairness)": "{\"description\": \"min=0.126, mean=0.138, max=0.155, sum=0.414 (3)\", \"tab\": \"Fairness\", \"score\": \"0.1378972222222222\"}", "MS MARCO (TREC) - NDCG@10 (Fairness)": "{\"description\": \"min=0.296, mean=0.328, max=0.372, sum=0.985 (3)\", \"tab\": \"Fairness\", \"score\": \"0.3284974893691146\"}", "MS MARCO (regular) - Denoised inference time (s)": "{\"description\": \"min=0.415, mean=0.428, max=0.44, sum=1.283 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.4278073636067708\"}", "MS MARCO (TREC) - Denoised inference time (s)": "{\"description\": \"min=0.412, mean=0.424, max=0.437, sum=1.272 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.42392066375968995\"}", "MS MARCO (regular) - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "MS MARCO (regular) - # train": "{\"description\": \"min=2, mean=2, max=2, sum=6 (3)\", \"tab\": \"General information\", \"score\": \"2.0\"}", "MS MARCO (regular) - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "MS MARCO (regular) - # prompt tokens": "{\"description\": \"min=349.303, mean=385.636, max=423.303, sum=1156.909 (3)\", \"tab\": \"General information\", \"score\": \"385.63633333333337\"}", "MS MARCO (regular) - # output tokens": "{\"description\": \"min=2.004, mean=2.011, max=2.023, sum=6.034 (3)\", \"tab\": \"General information\", \"score\": \"2.0113333333333334\"}", "MS MARCO (regular) - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "MS MARCO (TREC) - # eval": "{\"description\": \"min=43, mean=43, max=43, sum=129 (3)\", \"tab\": \"General information\", \"score\": \"43.0\"}", "MS MARCO (TREC) - # train": "{\"description\": \"min=2, mean=2, max=2, sum=6 (3)\", \"tab\": \"General information\", \"score\": \"2.0\"}", "MS MARCO (TREC) - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "MS MARCO (TREC) - # prompt tokens": "{\"description\": \"min=337.047, mean=373.38, max=411.047, sum=1120.14 (3)\", \"tab\": \"General information\", \"score\": \"373.3798449612403\"}", "MS MARCO (TREC) - # output tokens": "{\"description\": \"min=2.023, mean=2.023, max=2.023, sum=6.07 (3)\", \"tab\": \"General information\", \"score\": \"2.0232558139534884\"}", "MS MARCO (TREC) - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "MS MARCO (regular) - Stereotypes (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (regular) - Stereotypes (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (regular) - Representation (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (regular) - Representation (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (TREC) - Stereotypes (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (TREC) - Stereotypes (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (TREC) - Representation (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (TREC) - Representation (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "MS MARCO (regular) - Toxic fraction": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Toxicity\", \"score\": \"\"}", "MS MARCO (TREC) - Toxic fraction": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Toxicity\", \"score\": \"\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#ms_marco_trec#ndcg__k_10", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MS MARCO (TREC)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MS MARCO (TREC)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "slice_key": null, "slice_name": null, "metric_name": "NDCG", "metric_id": "ndcg", "metric_key": "ndcg", "metric_source": "metric_config", "display_name": "NDCG", "canonical_display_name": "MS MARCO (TREC) / NDCG", "raw_evaluation_name": "MS MARCO (TREC)", "is_summary_score": false } }, { "evaluation_name": "CNN/DailyMail", "source_data": { "dataset_name": "CNN/DailyMail", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "ROUGE-2 on CNN/DailyMail", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "score_details": { "score": 0.143, "details": { "description": "min=0.127, mean=0.143, max=0.163, sum=0.859 (6)", "tab": "Accuracy", "CNN/DailyMail - Denoised inference time (s)": "{\"description\": \"min=1.956, mean=2.074, max=2.263, sum=12.445 (6)\", \"tab\": \"Efficiency\", \"score\": \"2.074164002425339\"}", "CNN/DailyMail - # eval": "{\"description\": \"min=466, mean=466, max=466, sum=2796 (6)\", \"tab\": \"General information\", \"score\": \"466.0\"}", "CNN/DailyMail - # train": "{\"description\": \"min=5, mean=5, max=5, sum=30 (6)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "CNN/DailyMail - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (6)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "CNN/DailyMail - # prompt tokens": "{\"description\": \"min=1203.032, mean=1213.032, max=1224.032, sum=7278.193 (6)\", \"tab\": \"General information\", \"score\": \"1213.0321888412018\"}", "CNN/DailyMail - # output tokens": "{\"description\": \"min=61.569, mean=67.049, max=76.034, sum=402.296 (6)\", \"tab\": \"General information\", \"score\": \"67.04935622317596\"}", "CNN/DailyMail - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=18 (6)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "CNN/DailyMail - Stereotypes (race)": "{\"description\": \"min=0.608, mean=0.633, max=0.647, sum=3.801 (6)\", \"tab\": \"Bias\", \"score\": \"0.6334968330766649\"}", "CNN/DailyMail - Stereotypes (gender)": "{\"description\": \"min=0.39, mean=0.4, max=0.407, sum=2.398 (6)\", \"tab\": \"Bias\", \"score\": \"0.39959768497778553\"}", "CNN/DailyMail - Representation (race)": "{\"description\": \"min=0.263, mean=0.351, max=0.399, sum=2.104 (6)\", \"tab\": \"Bias\", \"score\": \"0.3506178570090534\"}", "CNN/DailyMail - Representation (gender)": "{\"description\": \"min=0.115, mean=0.13, max=0.14, sum=0.782 (6)\", \"tab\": \"Bias\", \"score\": \"0.1303299541894603\"}", "CNN/DailyMail - Toxic fraction": "{\"description\": \"min=0, mean=0.001, max=0.002, sum=0.009 (6)\", \"tab\": \"Toxicity\", \"score\": \"0.001430615164520744\"}", "CNN/DailyMail - SummaC": "{\"description\": \"min=0.514, mean=0.539, max=0.586, sum=1.617 (3)\", \"tab\": \"Summarization metrics\", \"score\": \"0.5391092885196874\"}", "CNN/DailyMail - QAFactEval": "{\"description\": \"min=4.706, mean=4.81, max=4.896, sum=28.859 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"4.809910581145076\"}", "CNN/DailyMail - BERTScore (F1)": "{\"description\": \"min=0.247, mean=0.275, max=0.302, sum=0.824 (3)\", \"tab\": \"Summarization metrics\", \"score\": \"0.2747429286177279\"}", "CNN/DailyMail - Coverage": "{\"description\": \"min=0.966, mean=0.973, max=0.984, sum=5.84 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"0.9733042514029583\"}", "CNN/DailyMail - Density": "{\"description\": \"min=31.118, mean=41.027, max=60.066, sum=246.163 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"41.02711755812993\"}", "CNN/DailyMail - Compression": "{\"description\": \"min=8.092, mean=9.888, max=11.258, sum=59.326 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"9.887609814491976\"}", "CNN/DailyMail - HumanEval-faithfulness": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}", "CNN/DailyMail - HumanEval-relevance": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}", "CNN/DailyMail - HumanEval-coherence": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#cnn_dailymail#rouge_2__n_2", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CNN/DailyMail", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CNN/DailyMail", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "CNN/DailyMail / ROUGE-2", "raw_evaluation_name": "CNN/DailyMail", "is_summary_score": false } }, { "evaluation_name": "XSUM", "source_data": { "dataset_name": "XSUM", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "ROUGE-2 on XSUM", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "score_details": { "score": 0.122, "details": { "description": "min=0.118, mean=0.122, max=0.127, sum=0.733 (6)", "tab": "Accuracy", "XSUM - Denoised inference time (s)": "{\"description\": \"min=1.055, mean=1.07, max=1.082, sum=6.42 (6)\", \"tab\": \"Efficiency\", \"score\": \"1.0700079645773009\"}", "XSUM - # eval": "{\"description\": \"min=518, mean=518, max=518, sum=3108 (6)\", \"tab\": \"General information\", \"score\": \"518.0\"}", "XSUM - # train": "{\"description\": \"min=5, mean=5, max=5, sum=30 (6)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "XSUM - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (6)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "XSUM - # prompt tokens": "{\"description\": \"min=1099.388, mean=1133.388, max=1172.388, sum=6800.328 (6)\", \"tab\": \"General information\", \"score\": \"1133.388030888031\"}", "XSUM - # output tokens": "{\"description\": \"min=19.975, mean=20.468, max=21.141, sum=122.807 (6)\", \"tab\": \"General information\", \"score\": \"20.467824967824967\"}", "XSUM - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=18 (6)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "XSUM - Stereotypes (race)": "{\"description\": \"min=0.667, mean=0.667, max=0.667, sum=4 (6)\", \"tab\": \"Bias\", \"score\": \"0.6666666666666666\"}", "XSUM - Stereotypes (gender)": "{\"description\": \"min=0.417, mean=0.442, max=0.485, sum=2.652 (6)\", \"tab\": \"Bias\", \"score\": \"0.44203142536475876\"}", "XSUM - Representation (race)": "{\"description\": \"min=0.439, mean=0.557, max=0.667, sum=3.34 (6)\", \"tab\": \"Bias\", \"score\": \"0.5566296694116243\"}", "XSUM - Representation (gender)": "{\"description\": \"min=0.149, mean=0.171, max=0.211, sum=1.025 (6)\", \"tab\": \"Bias\", \"score\": \"0.17086307216738958\"}", "XSUM - Toxic fraction": "{\"description\": \"min=0, mean=0.002, max=0.004, sum=0.012 (6)\", \"tab\": \"Toxicity\", \"score\": \"0.0019305019305019308\"}", "XSUM - SummaC": "{\"description\": \"min=-0.282, mean=-0.272, max=-0.264, sum=-0.815 (3)\", \"tab\": \"Summarization metrics\", \"score\": \"-0.2715132814883572\"}", "XSUM - QAFactEval": "{\"description\": \"min=3.221, mean=3.447, max=3.575, sum=20.68 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"3.446713620425662\"}", "XSUM - BERTScore (F1)": "{\"description\": \"min=0.424, mean=0.429, max=0.434, sum=1.287 (3)\", \"tab\": \"Summarization metrics\", \"score\": \"0.4288941077256343\"}", "XSUM - Coverage": "{\"description\": \"min=0.78, mean=0.783, max=0.785, sum=4.696 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"0.7826042118856411\"}", "XSUM - Density": "{\"description\": \"min=2.514, mean=2.64, max=2.767, sum=15.838 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"2.6397086455700927\"}", "XSUM - Compression": "{\"description\": \"min=18.382, mean=19.012, max=19.445, sum=114.069 (6)\", \"tab\": \"Summarization metrics\", \"score\": \"19.011567725134377\"}", "XSUM - HumanEval-faithfulness": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}", "XSUM - HumanEval-relevance": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}", "XSUM - HumanEval-coherence": "{\"description\": \"2 matching runs, but no matching metrics\", \"tab\": \"Summarization metrics\", \"score\": \"\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#xsum#rouge_2__n_2", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "XSUM", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "XSUM", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "XSUM / ROUGE-2", "raw_evaluation_name": "XSUM", "is_summary_score": false } }, { "evaluation_name": "IMDB", "source_data": { "dataset_name": "IMDB", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on IMDB", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.953, "details": { "description": "min=0.947, mean=0.953, max=0.957, sum=2.859 (3)", "tab": "Accuracy", "IMDB - ECE (10-bin)": "{\"description\": \"min=0.152, mean=0.158, max=0.166, sum=0.473 (3)\", \"tab\": \"Calibration\", \"score\": \"0.15775206410447826\"}", "IMDB - EM (Robustness)": "{\"description\": \"min=0.932, mean=0.941, max=0.948, sum=2.822 (3)\", \"tab\": \"Robustness\", \"score\": \"0.9406666666666667\"}", "IMDB - EM (Fairness)": "{\"description\": \"min=0.94, mean=0.946, max=0.95, sum=2.839 (3)\", \"tab\": \"Fairness\", \"score\": \"0.9463333333333331\"}", "IMDB - Denoised inference time (s)": "{\"description\": \"min=0.59, mean=0.732, max=0.881, sum=2.197 (3)\", \"tab\": \"Efficiency\", \"score\": \"0.7321998525390631\"}", "IMDB - # eval": "{\"description\": \"min=1000, mean=1000, max=1000, sum=3000 (3)\", \"tab\": \"General information\", \"score\": \"1000.0\"}", "IMDB - # train": "{\"description\": \"min=4.915, mean=4.972, max=5, sum=14.915 (3)\", \"tab\": \"General information\", \"score\": \"4.971666666666667\"}", "IMDB - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (3)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "IMDB - # prompt tokens": "{\"description\": \"min=853.851, mean=1281.577, max=1725.03, sum=3844.732 (3)\", \"tab\": \"General information\", \"score\": \"1281.5773333333334\"}", "IMDB - # output tokens": "{\"description\": \"min=2, mean=2, max=2, sum=6 (3)\", \"tab\": \"General information\", \"score\": \"2.0\"}", "IMDB - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=9 (3)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "IMDB - Stereotypes (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "IMDB - Stereotypes (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "IMDB - Representation (race)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "IMDB - Representation (gender)": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Bias\", \"score\": \"\"}", "IMDB - Toxic fraction": "{\"description\": \"1 matching runs, but no matching metrics\", \"tab\": \"Toxicity\", \"score\": \"\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#imdb#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "IMDB", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "IMDB", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "IMDB / Exact Match", "raw_evaluation_name": "IMDB", "is_summary_score": false } }, { "evaluation_name": "CivilComments", "source_data": { "dataset_name": "CivilComments", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on CivilComments", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.529, "details": { "description": "min=0.014, mean=0.529, max=0.991, sum=28.55 (54)", "tab": "Accuracy", "CivilComments - ECE (10-bin)": "{\"description\": \"min=0.228, mean=0.408, max=0.593, sum=22.008 (54)\", \"tab\": \"Calibration\", \"score\": \"0.4075612338805137\"}", "CivilComments - EM (Robustness)": "{\"description\": \"min=0.014, mean=0.417, max=0.938, sum=22.51 (54)\", \"tab\": \"Robustness\", \"score\": \"0.41686056018907397\"}", "CivilComments - EM (Fairness)": "{\"description\": \"min=0.014, mean=0.482, max=0.962, sum=26.023 (54)\", \"tab\": \"Fairness\", \"score\": \"0.4819034071645267\"}", "CivilComments - Denoised inference time (s)": "{\"description\": \"min=0.418, mean=0.482, max=0.621, sum=26.002 (54)\", \"tab\": \"Efficiency\", \"score\": \"0.48152748003997736\"}", "CivilComments - # eval": "{\"description\": \"min=74, mean=371.556, max=683, sum=20064 (54)\", \"tab\": \"General information\", \"score\": \"371.55555555555554\"}", "CivilComments - # train": "{\"description\": \"min=5, mean=5, max=5, sum=270 (54)\", \"tab\": \"General information\", \"score\": \"5.0\"}", "CivilComments - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (54)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "CivilComments - # prompt tokens": "{\"description\": \"min=271.927, mean=532.602, max=942.498, sum=28760.487 (54)\", \"tab\": \"General information\", \"score\": \"532.6016121330534\"}", "CivilComments - # output tokens": "{\"description\": \"min=2, mean=2, max=2, sum=108 (54)\", \"tab\": \"General information\", \"score\": \"2.0\"}", "CivilComments - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=162 (54)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "CivilComments - Stereotypes (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "CivilComments - Stereotypes (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "CivilComments - Representation (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "CivilComments - Representation (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "CivilComments - Toxic fraction": "{\"description\": \"min=0, mean=0, max=0, sum=0 (54)\", \"tab\": \"Toxicity\", \"score\": \"0.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#civilcomments#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CivilComments", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CivilComments", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "CivilComments / Exact Match", "raw_evaluation_name": "CivilComments", "is_summary_score": false } }, { "evaluation_name": "RAFT", "source_data": { "dataset_name": "RAFT", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "metric_config": { "evaluation_description": "EM on RAFT", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "score_details": { "score": 0.658, "details": { "description": "min=0.2, mean=0.658, max=0.975, sum=21.7 (33)", "tab": "Accuracy", "RAFT - ECE (10-bin)": "{\"description\": \"min=0.113, mean=0.244, max=0.466, sum=8.048 (33)\", \"tab\": \"Calibration\", \"score\": \"0.24386423436086976\"}", "RAFT - EM (Robustness)": "{\"description\": \"min=0.025, mean=0.513, max=0.775, sum=16.925 (33)\", \"tab\": \"Robustness\", \"score\": \"0.5128787878787878\"}", "RAFT - EM (Fairness)": "{\"description\": \"min=0.175, mean=0.636, max=0.975, sum=21 (33)\", \"tab\": \"Fairness\", \"score\": \"0.6363636363636364\"}", "RAFT - Denoised inference time (s)": "{\"description\": \"min=0.401, mean=0.59, max=0.888, sum=19.483 (33)\", \"tab\": \"Efficiency\", \"score\": \"0.5903971827651516\"}", "RAFT - # eval": "{\"description\": \"min=40, mean=40, max=40, sum=1320 (33)\", \"tab\": \"General information\", \"score\": \"40.0\"}", "RAFT - # train": "{\"description\": \"min=0.95, mean=4.658, max=5, sum=153.7 (33)\", \"tab\": \"General information\", \"score\": \"4.657575757575757\"}", "RAFT - truncated": "{\"description\": \"min=0, mean=0, max=0, sum=0 (33)\", \"tab\": \"General information\", \"score\": \"0.0\"}", "RAFT - # prompt tokens": "{\"description\": \"min=212.25, mean=712.248, max=1745.25, sum=23504.175 (33)\", \"tab\": \"General information\", \"score\": \"712.2477272727273\"}", "RAFT - # output tokens": "{\"description\": \"min=1.95, mean=3.59, max=6.575, sum=118.475 (33)\", \"tab\": \"General information\", \"score\": \"3.590151515151515\"}", "RAFT - # trials": "{\"description\": \"min=3, mean=3, max=3, sum=99 (33)\", \"tab\": \"General information\", \"score\": \"3.0\"}", "RAFT - Stereotypes (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "RAFT - Stereotypes (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "RAFT - Representation (race)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "RAFT - Representation (gender)": "{\"description\": \"(0)\", \"tab\": \"Bias\", \"score\": \"\"}", "RAFT - Toxic fraction": "{\"description\": \"min=0, mean=0, max=0, sum=0 (33)\", \"tab\": \"Toxicity\", \"score\": \"0.0\"}" } }, "generation_config": { "additional_details": {} }, "evaluation_result_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228#raft#exact_match", "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "RAFT", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "RAFT", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "RAFT / Exact Match", "raw_evaluation_name": "RAFT", "is_summary_score": false } } ], "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "instance_level_data": null, "eval_summary_ids": [ "helm_classic", "helm_classic_boolq", "helm_classic_civilcomments", "helm_classic_cnn_dailymail", "helm_classic_hellaswag", "helm_classic_imdb", "helm_classic_mmlu", "helm_classic_ms_marco_trec", "helm_classic_narrativeqa", "helm_classic_naturalquestions_open_book", "helm_classic_openbookqa", "helm_classic_quac", "helm_classic_raft", "helm_classic_truthfulqa", "helm_classic_xsum" ] } ] }, "evaluation_summaries_by_category": { "knowledge": [ { "eval_summary_id": "helm_classic", "benchmark": "Helm classic", "benchmark_family_key": "helm_classic", "benchmark_family_name": "Helm classic", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "Helm classic", "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "Helm classic", "display_name": "Helm classic", "canonical_display_name": "Helm classic", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Win Rate" ], "primary_metric_name": "Win Rate", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_win_rate", "legacy_eval_summary_id": "helm_classic_helm_classic", "evaluation_name": "helm_classic", "display_name": "Helm classic / Win Rate", "canonical_display_name": "Helm classic / Win Rate", "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Win Rate", "metric_id": "win_rate", "metric_key": "win_rate", "metric_source": "metric_config", "metric_config": { "evaluation_description": "How many models this model outperform on average (over columns).", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_evaluation_name": "Mean win rate" }, "metric_id": "win_rate", "metric_name": "Win Rate", "metric_kind": "win_rate", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.433, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.433, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "Helm classic", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "Helm classic", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "slice_key": null, "slice_name": null, "metric_name": "Win Rate", "metric_id": "win_rate", "metric_key": "win_rate", "metric_source": "metric_config", "display_name": "Win Rate", "canonical_display_name": "Helm classic / Win Rate", "raw_evaluation_name": "helm_classic", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.433, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_imdb", "benchmark": "IMDB", "benchmark_family_key": "helm_classic", "benchmark_family_name": "IMDB", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "IMDB", "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "IMDB", "display_name": "IMDB", "canonical_display_name": "IMDB", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "IMDB", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_imdb_exact_match", "legacy_eval_summary_id": "helm_classic_imdb", "evaluation_name": "IMDB", "display_name": "IMDB / Exact Match", "canonical_display_name": "IMDB / Exact Match", "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on IMDB", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.953, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.953, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "IMDB", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "IMDB", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "IMDB / Exact Match", "raw_evaluation_name": "IMDB", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.953, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_mmlu", "benchmark": "MMLU", "benchmark_family_key": "helm_classic", "benchmark_family_name": "MMLU", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MMLU", "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "MMLU", "display_name": "MMLU", "canonical_display_name": "MMLU", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "MMLU", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "Measuring Massive Multitask Language Understanding (MMLU)", "overview": "MMLU is a multiple-choice question-answering benchmark that measures a text model's multitask accuracy across 57 distinct tasks. It is designed to test a wide range of knowledge and problem-solving abilities, covering diverse academic and professional subjects from elementary to advanced levels.", "benchmark_type": "single", "appears_in": [ "helm_classic", "helm_lite" ], "data_type": "text", "domains": [ "STEM", "humanities", "social sciences" ], "languages": [ "English" ], "similar_benchmarks": [ "GLUE", "SuperGLUE" ], "resources": [ "https://arxiv.org/abs/2009.03300", "https://huggingface.co/datasets/cais/mmlu", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To bridge the gap between the wide-ranging knowledge models acquire during pretraining and existing evaluation measures by assessing models across a diverse set of academic and professional subjects.", "audience": [ "Researchers analyzing model capabilities and identifying shortcomings" ], "tasks": [ "Multiple-choice question answering" ], "limitations": "Models exhibit lopsided performance, frequently do not know when they are wrong, and have near-random accuracy on some socially important subjects like morality and law.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The dataset is an original source with expert-generated questions.", "size": "The dataset contains over 100,000 examples, with a test split of 14,042 examples, a validation split of 1,531 examples, a dev split of 285 examples, and an auxiliary training split of 99,842 examples.", "format": "parquet", "annotation": "The dataset has no additional annotations; each question provides the correct answer as a class label (A, B, C, or D)." }, "methodology": { "methods": [ "Models are evaluated exclusively in zero-shot and few-shot settings to measure knowledge acquired during pretraining." ], "metrics": [ "MMLU (accuracy)" ], "calculation": "The overall score is an average accuracy across the 57 tasks.", "interpretation": "Higher scores indicate better performance. Near random-chance accuracy indicates weak performance. The very largest GPT-3 model improved over random chance by almost 20 percentage points on average, but models still need substantial improvements to reach expert-level accuracy.", "baseline_results": "Paper baselines: Most recent models have near random-chance accuracy. The very largest GPT-3 model improved over random chance by almost 20 percentage points on average. EEE results: Yi 34B scored 0.6500, Anthropic-LM v4-s3 52B scored 0.4810. The mean score across 2 evaluated models is 0.5655.", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "MIT License", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "purpose_and_intended_users.out_of_scope_uses", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T13:14:49.605975", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "STEM", "humanities", "social sciences" ], "languages": [ "English" ], "tasks": [ "Multiple-choice question answering" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_mmlu_exact_match", "legacy_eval_summary_id": "helm_classic_mmlu", "evaluation_name": "MMLU", "display_name": "MMLU / Exact Match", "canonical_display_name": "MMLU / Exact Match", "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on MMLU", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.27, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.27, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MMLU", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MMLU", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "MMLU / Exact Match", "raw_evaluation_name": "MMLU", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.27, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_ms_marco_trec", "benchmark": "MS MARCO (TREC)", "benchmark_family_key": "helm_classic", "benchmark_family_name": "MS MARCO (TREC)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MS MARCO (TREC)", "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "MS MARCO (TREC)", "display_name": "MS MARCO (TREC)", "canonical_display_name": "MS MARCO (TREC)", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "MS MARCO (TREC)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "NDCG" ], "primary_metric_name": "NDCG", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_ms_marco_trec_ndcg", "legacy_eval_summary_id": "helm_classic_ms_marco_trec", "evaluation_name": "MS MARCO (TREC)", "display_name": "MS MARCO (TREC) / NDCG", "canonical_display_name": "MS MARCO (TREC) / NDCG", "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "NDCG", "metric_id": "ndcg", "metric_key": "ndcg", "metric_source": "metric_config", "metric_config": { "evaluation_description": "NDCG@10 on MS MARCO (TREC)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "ndcg", "metric_name": "NDCG@10", "metric_kind": "ndcg", "metric_unit": "proportion", "metric_parameters": { "k": 10 } }, "models_count": 1, "top_score": 0.341, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.341, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MS MARCO (TREC)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MS MARCO (TREC)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "slice_key": null, "slice_name": null, "metric_name": "NDCG", "metric_id": "ndcg", "metric_key": "ndcg", "metric_source": "metric_config", "display_name": "NDCG", "canonical_display_name": "MS MARCO (TREC) / NDCG", "raw_evaluation_name": "MS MARCO (TREC)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.341, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_narrativeqa", "benchmark": "NarrativeQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NarrativeQA", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA", "canonical_display_name": "NarrativeQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "NarrativeQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_narrativeqa_f1", "legacy_eval_summary_id": "helm_classic_narrativeqa", "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA / F1", "canonical_display_name": "NarrativeQA / F1", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on NarrativeQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.672, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.672, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.672, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_naturalquestions_open_book", "benchmark": "NaturalQuestions (open-book)", "benchmark_family_key": "helm_classic", "benchmark_family_name": "NaturalQuestions (open-book)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NaturalQuestions (open-book)", "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "NaturalQuestions (open-book)", "display_name": "NaturalQuestions (open-book)", "canonical_display_name": "NaturalQuestions (open-book)", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "NaturalQuestions (open-book)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_naturalquestions_open_book_f1", "legacy_eval_summary_id": "helm_classic_naturalquestions_open_book", "evaluation_name": "NaturalQuestions (open-book)", "display_name": "NaturalQuestions (open-book) / F1", "canonical_display_name": "NaturalQuestions (open-book) / F1", "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on NaturalQuestions (open-book)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.578, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.578, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NaturalQuestions (open-book)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NaturalQuestions (open-book)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NaturalQuestions (open-book) / F1", "raw_evaluation_name": "NaturalQuestions (open-book)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.578, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_openbookqa", "benchmark": "OpenbookQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "OpenbookQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "OpenbookQA", "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "OpenbookQA", "display_name": "OpenbookQA", "canonical_display_name": "OpenbookQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "OpenbookQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_openbookqa_exact_match", "legacy_eval_summary_id": "helm_classic_openbookqa", "evaluation_name": "OpenbookQA", "display_name": "OpenbookQA / Exact Match", "canonical_display_name": "OpenbookQA / Exact Match", "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on OpenbookQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.52, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.52, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "OpenbookQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "OpenbookQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "OpenbookQA / Exact Match", "raw_evaluation_name": "OpenbookQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.52, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_quac", "benchmark": "QuAC", "benchmark_family_key": "helm_classic", "benchmark_family_name": "QuAC", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "QuAC", "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "QuAC", "display_name": "QuAC", "canonical_display_name": "QuAC", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "QuAC", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "QuAC", "overview": "QuAC (Question Answering in Context) is a benchmark that measures a model's ability to answer questions within an information-seeking dialogue. It contains 14,000 dialogues comprising 100,000 question-answer pairs. The dataset is distinctive because questions are often open-ended, context-dependent, unanswerable, or only meaningful within the dialog flow, presenting challenges not found in standard machine comprehension datasets.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "question answering", "dialogue modeling", "text generation" ], "languages": [ "English" ], "similar_benchmarks": [ "SQuAD" ], "resources": [ "http://quac.ai", "https://arxiv.org/abs/1808.07036", "https://huggingface.co/datasets/allenai/quac", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To enable models to learn from and participate in information-seeking dialog, handling context-dependent, elliptical, and sometimes unanswerable questions.", "audience": [ "Not specified" ], "tasks": [ "Extractive question answering", "Text generation", "Fill mask" ], "limitations": "Some questions have lower quality annotations; the dataset filters out the noisiest ~10% of annotations where human F1 is below 40. Questions can be open-ended, unanswerable, or only meaningful within the dialog context, posing inherent challenges.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The data is crowdsourced via an interactive dialog between two crowd workers: one acting as a student asking questions to learn about a hidden Wikipedia text, and the other acting as a teacher who answers using short excerpts from that text. The source data comes from Wikipedia.", "size": "The dataset contains 98,407 question-answer pairs from 13,594 dialogs, based on 8,854 unique sections from 3,611 unique Wikipedia articles. The training set has 83,568 questions (11,567 dialogs), the validation set has 7,354 questions (1,000 dialogs), and the test set has 7,353 questions (1,002 dialogs). The dataset size is between 10,000 and 100,000 examples.", "format": "Each dialog is a sequence of question-answer pairs centered around a Wikipedia section. The teacher's response includes a text span, a 'yes/no' indication, a 'no answer' indication, and an encouragement for follow-up questions.", "annotation": "Questions are answered by a teacher selecting short excerpts (spans) from the Wikipedia text. The training set has one reference answer per question, while the validation and test sets each have five reference answers per question to improve evaluation reliability. For evaluation, questions with a human F1 score lower than 40 are not used, as manual inspection revealed lower quality below this threshold." }, "methodology": { "methods": [ "Models predict a text span to answer a question about a Wikipedia section, given a dialog history of previous questions and answers.", "The evaluation uses a reading comprehension architecture extended to model dialog context." ], "metrics": [ "Word-level F1" ], "calculation": "Precision and recall are computed over overlapping words after removing stopwords. For 'no answer' questions, F1 is 1 if correctly predicted and 0 otherwise. The maximum F1 among all references is computed for each question.", "interpretation": "Higher F1 scores indicate better performance. The best model underperforms humans by 20 F1, indicating significant room for improvement.", "baseline_results": "Paper baselines: The best model underperforms humans by 20 F1, but specific model names and scores are not provided. EEE results: Anthropic-LM v4-s3 52B achieves an F1 score of 0.431.", "validation": "Quality assurance includes using multiple references for development and test questions, filtering out questions with low human F1 scores, and manual inspection of low-quality annotations." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "MIT License", "consent_procedures": "Dialogs were created by two crowd workers, but the specific compensation or platform details are not provided in the paper.", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "purpose_and_intended_users.audience", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T13:45:24.009083", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "question answering", "dialogue modeling", "text generation" ], "languages": [ "English" ], "tasks": [ "Extractive question answering", "Text generation", "Fill mask" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_quac_f1", "legacy_eval_summary_id": "helm_classic_quac", "evaluation_name": "QuAC", "display_name": "QuAC / F1", "canonical_display_name": "QuAC / F1", "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on QuAC", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.362, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.362, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "QuAC", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "QuAC", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "QuAC / F1", "raw_evaluation_name": "QuAC", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.362, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_raft", "benchmark": "RAFT", "benchmark_family_key": "helm_classic", "benchmark_family_name": "RAFT", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "RAFT", "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "RAFT", "display_name": "RAFT", "canonical_display_name": "RAFT", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "RAFT", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_raft_exact_match", "legacy_eval_summary_id": "helm_classic_raft", "evaluation_name": "RAFT", "display_name": "RAFT / Exact Match", "canonical_display_name": "RAFT / Exact Match", "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on RAFT", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.658, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.658, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "RAFT", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "RAFT", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "RAFT / Exact Match", "raw_evaluation_name": "RAFT", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.658, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_truthfulqa", "benchmark": "TruthfulQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "TruthfulQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "TruthfulQA", "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "TruthfulQA", "display_name": "TruthfulQA", "canonical_display_name": "TruthfulQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "TruthfulQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_truthfulqa_exact_match", "legacy_eval_summary_id": "helm_classic_truthfulqa", "evaluation_name": "TruthfulQA", "display_name": "TruthfulQA / Exact Match", "canonical_display_name": "TruthfulQA / Exact Match", "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on TruthfulQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.193, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.193, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "TruthfulQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "TruthfulQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "TruthfulQA / Exact Match", "raw_evaluation_name": "TruthfulQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.193, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_xsum", "benchmark": "XSUM", "benchmark_family_key": "helm_classic", "benchmark_family_name": "XSUM", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "XSUM", "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "XSUM", "display_name": "XSUM", "canonical_display_name": "XSUM", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "XSUM", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "ROUGE-2" ], "primary_metric_name": "ROUGE-2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_xsum_rouge_2", "legacy_eval_summary_id": "helm_classic_xsum", "evaluation_name": "XSUM", "display_name": "XSUM / ROUGE-2", "canonical_display_name": "XSUM / ROUGE-2", "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ROUGE-2 on XSUM", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "models_count": 1, "top_score": 0.122, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.122, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "XSUM", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "XSUM", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "XSUM / ROUGE-2", "raw_evaluation_name": "XSUM", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.122, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "language_understanding": [ { "eval_summary_id": "helm_classic_boolq", "benchmark": "BoolQ", "benchmark_family_key": "helm_classic", "benchmark_family_name": "BoolQ", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "BoolQ", "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "BoolQ", "display_name": "BoolQ", "canonical_display_name": "BoolQ", "is_summary_score": false, "category": "language_understanding", "source_data": { "dataset_name": "BoolQ", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "BoolQ", "overview": "BoolQ is a benchmark that measures a model's ability to answer naturally occurring yes/no questions, framed as a reading comprehension task. The questions are generated in unprompted and unconstrained settings, often querying complex, non-factoid information and requiring difficult entailment-like inference. The dataset consists of a single task: answering yes/no questions given a supporting passage.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "natural language understanding", "reading comprehension", "natural language inference" ], "languages": [ "English" ], "similar_benchmarks": [ "MultiNLI", "SNLI", "QNLI", "SQuAD 2.0", "Natural Questions (NQ)", "QQP", "MS MARCO", "RACE", "bAbI stories" ], "resources": [ "https://arxiv.org/abs/1905.10044", "https://huggingface.co/datasets/google/boolq", "https://goo.gl/boolq", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To test models on their ability to answer naturally occurring yes/no questions, which are challenging and require complex inferential abilities beyond surface-level reasoning.", "audience": [ "Researchers in natural language understanding and reading comprehension" ], "tasks": [ "Yes/no question answering", "Text-pair classification" ], "limitations": "Annotation involved some errors and ambiguous cases. The use of singly-annotated examples is a trade-off for dataset size. Potential concerns about annotation artifacts are acknowledged.", "out_of_scope_uses": "The paper does not explicitly state what the benchmark is not designed for." }, "data": { "source": "The data consists of naturally occurring yes/no questions authored by people who were not prompted to write specific question types and did not know the answers. The passages are excerpts from sources like Wikipedia.", "size": "15,942 examples total, with 9,427 in the train split and 3,270 in the validation split. The dataset size category is between 10,000 and 100,000 examples.", "format": "parquet", "annotation": "Questions were answered by human annotators. A quality check on a subset showed the main annotation process achieved 90% accuracy against a gold-standard set labeled by three authors. The training, development, and test sets use singly-annotated examples." }, "methodology": { "methods": [ "Models are evaluated by fine-tuning on the BoolQ training set, potentially after transfer learning from other datasets or unsupervised pre-training. Zero-shot or direct use of pre-trained models without fine-tuning did not outperform the majority baseline.", "The task requires providing a yes/no (boolean) answer to a question based on a given passage." ], "metrics": [ "Accuracy" ], "calculation": "The overall score is the accuracy percentage on the test set.", "interpretation": "Higher accuracy indicates better performance. Human accuracy is 90%, and the majority baseline is approximately 62%.", "baseline_results": "Paper baselines: Majority baseline: 62.17% dev, 62.31% test; Recurrent model baseline: 69.6%; Best model (BERT large pre-trained on MultiNLI then fine-tuned on BoolQ): 80.4% accuracy; Human accuracy: 90%. EEE results: Anthropic-LM v4-s3 52B: 81.5%.", "validation": "Quality assurance involved author-led gold-standard annotation on a subset, showing 90% agreement. The development set was used for model selection, such as choosing the best model from five seeds based on its performance." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "cc-by-sa-3.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:08:51.830946", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "natural language understanding", "reading comprehension", "natural language inference" ], "languages": [ "English" ], "tasks": [ "Yes/no question answering", "Text-pair classification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_boolq_exact_match", "legacy_eval_summary_id": "helm_classic_boolq", "evaluation_name": "BoolQ", "display_name": "BoolQ / Exact Match", "canonical_display_name": "BoolQ / Exact Match", "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on BoolQ", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.722, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.722, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "BoolQ", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "BoolQ", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "BoolQ / Exact Match", "raw_evaluation_name": "BoolQ", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.722, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_cnn_dailymail", "benchmark": "CNN/DailyMail", "benchmark_family_key": "helm_classic", "benchmark_family_name": "CNN/DailyMail", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CNN/DailyMail", "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "CNN/DailyMail", "display_name": "CNN/DailyMail", "canonical_display_name": "CNN/DailyMail", "is_summary_score": false, "category": "language_understanding", "source_data": { "dataset_name": "CNN/DailyMail", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "CNN/DailyMail", "overview": "CNN/DailyMail is a benchmark for evaluating abstractive and extractive summarization models using news articles. It contains over 300,000 unique articles written by journalists from CNN and the Daily Mail. The dataset was originally created for machine reading and question answering, but later versions were restructured specifically for summarization tasks.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "summarization", "journalism", "news media" ], "languages": [ "English" ], "similar_benchmarks": "No facts provided about similar benchmarks.", "resources": [ "https://huggingface.co/datasets/abisee/cnn_dailymail", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To help develop models that can summarize long paragraphs of text into one or two sentences, aiding in the efficient presentation of information from large quantities of text.", "audience": [ "NLP researchers", "Summarization model developers" ], "tasks": [ "Summarization" ], "limitations": "News articles often place important information in the first third, which may affect summarization. A manual study found 25% of samples in an earlier version were difficult for humans due to ambiguity and coreference errors. Also, machine-generated summaries may differ in truth values from the original articles.", "out_of_scope_uses": "No facts provided about out-of-scope uses." }, "data": { "source": "The dataset consists of news articles and highlight sentences written by journalists at CNN and the Daily Mail. The CNN articles were collected from April 2007 to April 2015, and the Daily Mail articles from June 2010 to April 2015, sourced from archives on the Wayback Machine.", "size": "Over 300,000 unique articles, with 287,113 training examples, 13,368 validation examples, and 11,490 test examples.", "format": "parquet", "annotation": "The dataset does not contain additional annotations. The highlights are the original summaries written by the article authors and are used as the target for summarization." }, "methodology": { "methods": [ "Models generate a summary for a given news article, which is then compared to the author-written highlights." ], "metrics": [ "ROUGE-2" ], "calculation": "The ROUGE-2 score measures the overlap of bigrams between the generated summary and the reference highlights.", "interpretation": "Higher scores indicate better performance, as they reflect greater overlap with the reference summaries.", "baseline_results": "Paper baseline (Zhong et al., 2020): ROUGE-1 score of 44.41 for an extractive summarization model. Evaluation suite result (Anthropic-LM v4-s3 52B): ROUGE-2 score of 0.154.", "validation": "No facts provided about validation procedures." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "The dataset (version 3.0.0) is not anonymized, meaning individuals' names are present in the text.", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:15:47.316103", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "summarization", "journalism", "news media" ], "languages": [ "English" ], "tasks": [ "Summarization" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "ROUGE-2" ], "primary_metric_name": "ROUGE-2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_cnn_dailymail_rouge_2", "legacy_eval_summary_id": "helm_classic_cnn_dailymail", "evaluation_name": "CNN/DailyMail", "display_name": "CNN/DailyMail / ROUGE-2", "canonical_display_name": "CNN/DailyMail / ROUGE-2", "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ROUGE-2 on CNN/DailyMail", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "models_count": 1, "top_score": 0.143, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.143, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CNN/DailyMail", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CNN/DailyMail", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "CNN/DailyMail / ROUGE-2", "raw_evaluation_name": "CNN/DailyMail", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.143, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "reasoning": [ { "eval_summary_id": "helm_classic_hellaswag", "benchmark": "HellaSwag", "benchmark_family_key": "helm_classic", "benchmark_family_name": "HellaSwag", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "HellaSwag", "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "HellaSwag", "display_name": "HellaSwag", "canonical_display_name": "HellaSwag", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "HellaSwag", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "HellaSwag", "overview": "HellaSwag is a benchmark designed to measure commonsense natural language inference by testing a model's ability to select the most plausible follow-up event from four multiple-choice options. It is adversarially constructed to be challenging for state-of-the-art models, using a method called Adversarial Filtering to create difficult wrong answers that are obvious to humans but often misclassified by models.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "commonsense reasoning", "natural language inference" ], "languages": [ "English" ], "similar_benchmarks": [ "SWAG", "SNLI" ], "resources": [ "https://rowanzellers.com/hellaswag", "https://arxiv.org/abs/1905.07830", "https://huggingface.co/datasets/Rowan/hellaswag", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To create a challenge dataset that reveals the difficulty of commonsense inference for state-of-the-art models, demonstrating their lack of robustness and reliance on dataset biases rather than genuine reasoning. It aims to evaluate a model's ability to select the most plausible continuation of a given event description.", "audience": [ "NLP researchers" ], "tasks": [ "Four-way multiple-choice selection for event continuation", "Commonsense inference" ], "limitations": "The adversarial filtering process used to create the dataset, while effective at making it difficult for models, may also select examples where the ground truth answer is not the one preferred by human annotators, necessitating manual filtering to retain the best examples.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "Contexts are sourced from WikiHow instructional articles and ActivityNet video descriptions. Incorrect answer choices are generated by machines and then adversarially filtered.", "size": "The dataset contains 70,000 examples in total, with 5,001 in-domain validation examples and 5,000 zero-shot validation examples. The training set comprises 39,905 examples.", "format": "Parquet", "annotation": "Human crowd workers on Amazon Mechanical Turk validated the endings. They were presented with a context and six endings (one true, five machine-generated) and rated their plausibility. The process involved iterative filtering and replacement of unrealistic endings. Worker quality was ensured via an autograded test and fair pay. A gold standard check by three authors on a random sample showed 90% agreement with crowd annotations." }, "methodology": { "methods": [ "Models are evaluated via fine-tuning on the dataset.", "The benchmark also includes zero-shot evaluation on held-out categories." ], "metrics": [ "HellaSwag accuracy" ], "calculation": "The overall score is the accuracy percentage on the full validation or test sets. Performance is also broken down by subsets, such as in-domain versus zero-shot and by data source.", "interpretation": "Higher scores indicate better performance. Human performance is over 95%, which is considered strong. Model performance below 50% is reported, indicating a struggle, with a gap of over 45% from human performance on in-domain data.", "baseline_results": "Paper baselines: BERT-Large achieves 47.3% accuracy overall. ESIM + ELMo gets 33.3% accuracy. A BERT-Base model with a frozen encoder and an added LSTM performs 4.3% worse than fine-tuned BERT-Base. Evaluation suite results: Anthropic-LM v4-s3 52B achieves 0.807 (80.7%) accuracy.", "validation": "Human validation involved giving five crowd workers the same multiple-choice task and combining their answers via majority vote to establish a human performance baseline. The adversarial filtering process used iterative human ratings to ensure wrong answers were implausible." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Crowdworkers on Amazon Mechanical Turk participated voluntarily and were compensated, with pay described as fair. A qualification task was used to filter workers, and those who consistently preferred generated endings over real ones were disqualified.", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "baseline_results": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:47:07.561060", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "commonsense reasoning", "natural language inference" ], "languages": [ "English" ], "tasks": [ "Four-way multiple-choice selection for event continuation", "Commonsense inference" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_hellaswag_exact_match", "legacy_eval_summary_id": "helm_classic_hellaswag", "evaluation_name": "HellaSwag", "display_name": "HellaSwag / Exact Match", "canonical_display_name": "HellaSwag / Exact Match", "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on HellaSwag", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.739, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.739, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "HellaSwag", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "HellaSwag", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "HellaSwag / Exact Match", "raw_evaluation_name": "HellaSwag", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.739, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "safety": [ { "eval_summary_id": "helm_classic_civilcomments", "benchmark": "CivilComments", "benchmark_family_key": "helm_classic", "benchmark_family_name": "CivilComments", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CivilComments", "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "CivilComments", "display_name": "CivilComments", "canonical_display_name": "CivilComments", "is_summary_score": false, "category": "safety", "source_data": { "dataset_name": "CivilComments", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "CivilComments", "overview": "CivilComments is a benchmark designed to measure unintended identity-based bias in toxicity classification models. It uses a large, real-world dataset of online comments from the Civil Comments platform, extended with crowd-sourced annotations for toxicity and demographic identity references. This provides a nuanced evaluation of bias beyond synthetic datasets.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "tabular, text", "domains": [ "machine learning fairness", "bias measurement", "toxic comment classification", "text classification" ], "languages": [ "English" ], "similar_benchmarks": "The paper does not name other specific benchmark datasets, only referencing prior work using synthetic test sets.", "resources": [ "https://arxiv.org/abs/1903.04561", "https://huggingface.co/datasets/google/civil_comments", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To evaluate unintended identity-based bias in toxicity classification models using real data and nuanced metrics.", "audience": [ "Machine learning researchers and practitioners working on fairness, bias measurement, and mitigation, particularly in toxic comment classification." ], "tasks": [ "Binary toxicity classification (toxic vs. non-toxic)", "Analysis of performance across identity subgroups" ], "limitations": "The labeled set of identities is not comprehensive and does not provide universal coverage, representing a balance between coverage, annotator accuracy, and example count. The real-world data is potentially noisier than synthetic alternatives.", "out_of_scope_uses": [ "Developing effective strategies for choosing optimal thresholds to minimize bias" ] }, "data": { "source": "The data consists of online comments sourced from the Civil Comments platform, a commenting plugin for independent English-language news sites. The comments were publicly posted between 2015 and 2017.", "size": "The dataset contains approximately 1.8 million comments for training, with separate validation and test sets of approximately 97,320 examples each. All comments were labeled for toxicity, and a subset of 450,000 comments was additionally labeled for identity references.", "format": "parquet", "annotation": "Labeling was performed by crowd raters. Toxicity labels were applied using guidelines consistent with the Perspective API. For the identity-labeled subset, raters were shown comments and selected referenced identities (e.g., genders, races, ethnicities) from a provided list. Some comments for identity labeling were pre-selected by models to increase the frequency of identity content." }, "methodology": { "methods": [ "Models are evaluated by applying a suite of bias metrics to their predictions on the test set. The original paper demonstrates this using publicly accessible toxicity classifiers on the dataset." ], "metrics": [ "Subgroup AUC", "BPSN AUC", "BNSP AUC", "Negative Average Equality Gap (AEG)", "Positive Average Equality Gap (AEG)" ], "calculation": "The evaluation calculates five metrics for each identity subgroup to provide a multi-faceted view of bias. There is no single aggregated overall score.", "interpretation": "For the AUC metrics (Subgroup, BPSN, BNSP), higher values indicate better separability (fewer mis-orderings). For the Average Equality Gaps (Negative and Positive), lower values indicate better separability (more similar score distributions).", "baseline_results": "Paper baselines: Results for TOXICITY@1 and TOXICITY@6 from the Perspective API are reported, showing their Subgroup AUC, BPSN AUC, BNSP AUC, Negative AEG, and Positive AEG on a synthetic dataset for the lowest performing 20 subgroups. They are also compared on short comments within the human-labeled dataset for specific identities. EEE results: Anthropic-LM v4-s3 52B scored 0.6100 on the CivilComments metric.", "validation": "The evaluation assumes the human-provided labels are reliable. The identity labeling set was designed to balance coverage, crowd rater accuracy, and ensure sufficient examples per identity for meaningful results." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "The paper does not discuss how personally identifiable information (PII) in the online comments was handled or if data was anonymized.", "data_licensing": "Creative Commons Zero v1.0 Universal", "consent_procedures": "The paper does not describe compensation for crowdworkers or the specific platform used for annotation.", "compliance_with_regulations": "The paper does not mention IRB approval, GDPR compliance, or any other ethical review process." }, "possible_risks": [ { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Output bias", "description": [ "Generated content might unfairly represent certain groups or individuals." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/output-bias.html" } ], "flagged_fields": {}, "missing_fields": [], "card_info": { "created_at": "2026-03-17T12:38:43.250822", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "machine learning fairness", "bias measurement", "toxic comment classification", "text classification" ], "languages": [ "English" ], "tasks": [ "Binary toxicity classification (toxic vs. non-toxic)", "Analysis of performance across identity subgroups" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_civilcomments_exact_match", "legacy_eval_summary_id": "helm_classic_civilcomments", "evaluation_name": "CivilComments", "display_name": "CivilComments / Exact Match", "canonical_display_name": "CivilComments / Exact Match", "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on CivilComments", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.529, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.529, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CivilComments", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CivilComments", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "CivilComments / Exact Match", "raw_evaluation_name": "CivilComments", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.529, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "hierarchy_by_category": { "knowledge": [ { "eval_summary_id": "helm_classic", "benchmark": "Helm classic", "benchmark_family_key": "helm_classic", "benchmark_family_name": "Helm classic", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "Helm classic", "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "Helm classic", "display_name": "Helm classic", "canonical_display_name": "Helm classic", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Win Rate" ], "primary_metric_name": "Win Rate", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_win_rate", "legacy_eval_summary_id": "helm_classic_helm_classic", "evaluation_name": "helm_classic", "display_name": "Helm classic / Win Rate", "canonical_display_name": "Helm classic / Win Rate", "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Win Rate", "metric_id": "win_rate", "metric_key": "win_rate", "metric_source": "metric_config", "metric_config": { "evaluation_description": "How many models this model outperform on average (over columns).", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "additional_details": { "raw_evaluation_name": "Mean win rate" }, "metric_id": "win_rate", "metric_name": "Win Rate", "metric_kind": "win_rate", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.433, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.433, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "Helm classic", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "Helm classic", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "helm_classic", "benchmark_leaf_name": "Helm classic", "slice_key": null, "slice_name": null, "metric_name": "Win Rate", "metric_id": "win_rate", "metric_key": "win_rate", "metric_source": "metric_config", "display_name": "Win Rate", "canonical_display_name": "Helm classic / Win Rate", "raw_evaluation_name": "helm_classic", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.433, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_imdb", "benchmark": "IMDB", "benchmark_family_key": "helm_classic", "benchmark_family_name": "IMDB", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "IMDB", "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "IMDB", "display_name": "IMDB", "canonical_display_name": "IMDB", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "IMDB", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_imdb_exact_match", "legacy_eval_summary_id": "helm_classic_imdb", "evaluation_name": "IMDB", "display_name": "IMDB / Exact Match", "canonical_display_name": "IMDB / Exact Match", "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on IMDB", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.953, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.953, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "IMDB", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "IMDB", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "imdb", "benchmark_leaf_name": "IMDB", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "IMDB / Exact Match", "raw_evaluation_name": "IMDB", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.953, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_mmlu", "benchmark": "MMLU", "benchmark_family_key": "helm_classic", "benchmark_family_name": "MMLU", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MMLU", "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "MMLU", "display_name": "MMLU", "canonical_display_name": "MMLU", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "MMLU", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "Measuring Massive Multitask Language Understanding (MMLU)", "overview": "MMLU is a multiple-choice question-answering benchmark that measures a text model's multitask accuracy across 57 distinct tasks. It is designed to test a wide range of knowledge and problem-solving abilities, covering diverse academic and professional subjects from elementary to advanced levels.", "benchmark_type": "single", "appears_in": [ "helm_classic", "helm_lite" ], "data_type": "text", "domains": [ "STEM", "humanities", "social sciences" ], "languages": [ "English" ], "similar_benchmarks": [ "GLUE", "SuperGLUE" ], "resources": [ "https://arxiv.org/abs/2009.03300", "https://huggingface.co/datasets/cais/mmlu", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://storage.googleapis.com/crfm-helm-public/lite/benchmark_output/releases/v1.13.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To bridge the gap between the wide-ranging knowledge models acquire during pretraining and existing evaluation measures by assessing models across a diverse set of academic and professional subjects.", "audience": [ "Researchers analyzing model capabilities and identifying shortcomings" ], "tasks": [ "Multiple-choice question answering" ], "limitations": "Models exhibit lopsided performance, frequently do not know when they are wrong, and have near-random accuracy on some socially important subjects like morality and law.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The dataset is an original source with expert-generated questions.", "size": "The dataset contains over 100,000 examples, with a test split of 14,042 examples, a validation split of 1,531 examples, a dev split of 285 examples, and an auxiliary training split of 99,842 examples.", "format": "parquet", "annotation": "The dataset has no additional annotations; each question provides the correct answer as a class label (A, B, C, or D)." }, "methodology": { "methods": [ "Models are evaluated exclusively in zero-shot and few-shot settings to measure knowledge acquired during pretraining." ], "metrics": [ "MMLU (accuracy)" ], "calculation": "The overall score is an average accuracy across the 57 tasks.", "interpretation": "Higher scores indicate better performance. Near random-chance accuracy indicates weak performance. The very largest GPT-3 model improved over random chance by almost 20 percentage points on average, but models still need substantial improvements to reach expert-level accuracy.", "baseline_results": "Paper baselines: Most recent models have near random-chance accuracy. The very largest GPT-3 model improved over random chance by almost 20 percentage points on average. EEE results: Yi 34B scored 0.6500, Anthropic-LM v4-s3 52B scored 0.4810. The mean score across 2 evaluated models is 0.5655.", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "MIT License", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": {}, "missing_fields": [ "purpose_and_intended_users.out_of_scope_uses", "methodology.validation", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T13:14:49.605975", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "STEM", "humanities", "social sciences" ], "languages": [ "English" ], "tasks": [ "Multiple-choice question answering" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_mmlu_exact_match", "legacy_eval_summary_id": "helm_classic_mmlu", "evaluation_name": "MMLU", "display_name": "MMLU / Exact Match", "canonical_display_name": "MMLU / Exact Match", "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on MMLU", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.27, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.27, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MMLU", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MMLU", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "mmlu", "benchmark_leaf_name": "MMLU", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "MMLU / Exact Match", "raw_evaluation_name": "MMLU", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.27, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_ms_marco_trec", "benchmark": "MS MARCO (TREC)", "benchmark_family_key": "helm_classic", "benchmark_family_name": "MS MARCO (TREC)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MS MARCO (TREC)", "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "MS MARCO (TREC)", "display_name": "MS MARCO (TREC)", "canonical_display_name": "MS MARCO (TREC)", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "MS MARCO (TREC)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "NDCG" ], "primary_metric_name": "NDCG", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_ms_marco_trec_ndcg", "legacy_eval_summary_id": "helm_classic_ms_marco_trec", "evaluation_name": "MS MARCO (TREC)", "display_name": "MS MARCO (TREC) / NDCG", "canonical_display_name": "MS MARCO (TREC) / NDCG", "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "NDCG", "metric_id": "ndcg", "metric_key": "ndcg", "metric_source": "metric_config", "metric_config": { "evaluation_description": "NDCG@10 on MS MARCO (TREC)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "ndcg", "metric_name": "NDCG@10", "metric_kind": "ndcg", "metric_unit": "proportion", "metric_parameters": { "k": 10 } }, "models_count": 1, "top_score": 0.341, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.341, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "MS MARCO (TREC)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "MS MARCO (TREC)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "ms_marco_trec", "benchmark_leaf_name": "MS MARCO (TREC)", "slice_key": null, "slice_name": null, "metric_name": "NDCG", "metric_id": "ndcg", "metric_key": "ndcg", "metric_source": "metric_config", "display_name": "NDCG", "canonical_display_name": "MS MARCO (TREC) / NDCG", "raw_evaluation_name": "MS MARCO (TREC)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.341, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_narrativeqa", "benchmark": "NarrativeQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NarrativeQA", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA", "canonical_display_name": "NarrativeQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "NarrativeQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_narrativeqa_f1", "legacy_eval_summary_id": "helm_classic_narrativeqa", "evaluation_name": "NarrativeQA", "display_name": "NarrativeQA / F1", "canonical_display_name": "NarrativeQA / F1", "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on NarrativeQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.672, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.672, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NarrativeQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NarrativeQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "narrativeqa", "benchmark_leaf_name": "NarrativeQA", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NarrativeQA / F1", "raw_evaluation_name": "NarrativeQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.672, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_naturalquestions_open_book", "benchmark": "NaturalQuestions (open-book)", "benchmark_family_key": "helm_classic", "benchmark_family_name": "NaturalQuestions (open-book)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NaturalQuestions (open-book)", "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "NaturalQuestions (open-book)", "display_name": "NaturalQuestions (open-book)", "canonical_display_name": "NaturalQuestions (open-book)", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "NaturalQuestions (open-book)", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_naturalquestions_open_book_f1", "legacy_eval_summary_id": "helm_classic_naturalquestions_open_book", "evaluation_name": "NaturalQuestions (open-book)", "display_name": "NaturalQuestions (open-book) / F1", "canonical_display_name": "NaturalQuestions (open-book) / F1", "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on NaturalQuestions (open-book)", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.578, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.578, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "NaturalQuestions (open-book)", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "NaturalQuestions (open-book)", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "naturalquestions_open_book", "benchmark_leaf_name": "NaturalQuestions (open-book)", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "NaturalQuestions (open-book) / F1", "raw_evaluation_name": "NaturalQuestions (open-book)", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.578, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_openbookqa", "benchmark": "OpenbookQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "OpenbookQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "OpenbookQA", "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "OpenbookQA", "display_name": "OpenbookQA", "canonical_display_name": "OpenbookQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "OpenbookQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_openbookqa_exact_match", "legacy_eval_summary_id": "helm_classic_openbookqa", "evaluation_name": "OpenbookQA", "display_name": "OpenbookQA / Exact Match", "canonical_display_name": "OpenbookQA / Exact Match", "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on OpenbookQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.52, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.52, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "OpenbookQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "OpenbookQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "openbookqa", "benchmark_leaf_name": "OpenbookQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "OpenbookQA / Exact Match", "raw_evaluation_name": "OpenbookQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.52, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_quac", "benchmark": "QuAC", "benchmark_family_key": "helm_classic", "benchmark_family_name": "QuAC", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "QuAC", "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "QuAC", "display_name": "QuAC", "canonical_display_name": "QuAC", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "QuAC", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "QuAC", "overview": "QuAC (Question Answering in Context) is a benchmark that measures a model's ability to answer questions within an information-seeking dialogue. It contains 14,000 dialogues comprising 100,000 question-answer pairs. The dataset is distinctive because questions are often open-ended, context-dependent, unanswerable, or only meaningful within the dialog flow, presenting challenges not found in standard machine comprehension datasets.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "question answering", "dialogue modeling", "text generation" ], "languages": [ "English" ], "similar_benchmarks": [ "SQuAD" ], "resources": [ "http://quac.ai", "https://arxiv.org/abs/1808.07036", "https://huggingface.co/datasets/allenai/quac", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To enable models to learn from and participate in information-seeking dialog, handling context-dependent, elliptical, and sometimes unanswerable questions.", "audience": [ "Not specified" ], "tasks": [ "Extractive question answering", "Text generation", "Fill mask" ], "limitations": "Some questions have lower quality annotations; the dataset filters out the noisiest ~10% of annotations where human F1 is below 40. Questions can be open-ended, unanswerable, or only meaningful within the dialog context, posing inherent challenges.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The data is crowdsourced via an interactive dialog between two crowd workers: one acting as a student asking questions to learn about a hidden Wikipedia text, and the other acting as a teacher who answers using short excerpts from that text. The source data comes from Wikipedia.", "size": "The dataset contains 98,407 question-answer pairs from 13,594 dialogs, based on 8,854 unique sections from 3,611 unique Wikipedia articles. The training set has 83,568 questions (11,567 dialogs), the validation set has 7,354 questions (1,000 dialogs), and the test set has 7,353 questions (1,002 dialogs). The dataset size is between 10,000 and 100,000 examples.", "format": "Each dialog is a sequence of question-answer pairs centered around a Wikipedia section. The teacher's response includes a text span, a 'yes/no' indication, a 'no answer' indication, and an encouragement for follow-up questions.", "annotation": "Questions are answered by a teacher selecting short excerpts (spans) from the Wikipedia text. The training set has one reference answer per question, while the validation and test sets each have five reference answers per question to improve evaluation reliability. For evaluation, questions with a human F1 score lower than 40 are not used, as manual inspection revealed lower quality below this threshold." }, "methodology": { "methods": [ "Models predict a text span to answer a question about a Wikipedia section, given a dialog history of previous questions and answers.", "The evaluation uses a reading comprehension architecture extended to model dialog context." ], "metrics": [ "Word-level F1" ], "calculation": "Precision and recall are computed over overlapping words after removing stopwords. For 'no answer' questions, F1 is 1 if correctly predicted and 0 otherwise. The maximum F1 among all references is computed for each question.", "interpretation": "Higher F1 scores indicate better performance. The best model underperforms humans by 20 F1, indicating significant room for improvement.", "baseline_results": "Paper baselines: The best model underperforms humans by 20 F1, but specific model names and scores are not provided. EEE results: Anthropic-LM v4-s3 52B achieves an F1 score of 0.431.", "validation": "Quality assurance includes using multiple references for development and test questions, filtering out questions with low human F1 scores, and manual inspection of low-quality annotations." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "MIT License", "consent_procedures": "Dialogs were created by two crowd workers, but the specific compensation or platform details are not provided in the paper.", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "purpose_and_intended_users.audience", "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T13:45:24.009083", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "question answering", "dialogue modeling", "text generation" ], "languages": [ "English" ], "tasks": [ "Extractive question answering", "Text generation", "Fill mask" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "F1" ], "primary_metric_name": "F1", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_quac_f1", "legacy_eval_summary_id": "helm_classic_quac", "evaluation_name": "QuAC", "display_name": "QuAC / F1", "canonical_display_name": "QuAC / F1", "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "metric_config": { "evaluation_description": "F1 on QuAC", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "f1", "metric_name": "F1", "metric_kind": "f1", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.362, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.362, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "QuAC", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "QuAC", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "quac", "benchmark_leaf_name": "QuAC", "slice_key": null, "slice_name": null, "metric_name": "F1", "metric_id": "f1", "metric_key": "f1", "metric_source": "metric_config", "display_name": "F1", "canonical_display_name": "QuAC / F1", "raw_evaluation_name": "QuAC", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.362, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_raft", "benchmark": "RAFT", "benchmark_family_key": "helm_classic", "benchmark_family_name": "RAFT", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "RAFT", "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "RAFT", "display_name": "RAFT", "canonical_display_name": "RAFT", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "RAFT", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_raft_exact_match", "legacy_eval_summary_id": "helm_classic_raft", "evaluation_name": "RAFT", "display_name": "RAFT / Exact Match", "canonical_display_name": "RAFT / Exact Match", "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on RAFT", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.658, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.658, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "RAFT", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "RAFT", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "raft", "benchmark_leaf_name": "RAFT", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "RAFT / Exact Match", "raw_evaluation_name": "RAFT", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.658, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_truthfulqa", "benchmark": "TruthfulQA", "benchmark_family_key": "helm_classic", "benchmark_family_name": "TruthfulQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "TruthfulQA", "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "TruthfulQA", "display_name": "TruthfulQA", "canonical_display_name": "TruthfulQA", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "TruthfulQA", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_truthfulqa_exact_match", "legacy_eval_summary_id": "helm_classic_truthfulqa", "evaluation_name": "TruthfulQA", "display_name": "TruthfulQA / Exact Match", "canonical_display_name": "TruthfulQA / Exact Match", "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on TruthfulQA", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.193, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.193, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "TruthfulQA", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "TruthfulQA", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "truthfulqa", "benchmark_leaf_name": "TruthfulQA", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "TruthfulQA / Exact Match", "raw_evaluation_name": "TruthfulQA", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.193, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_xsum", "benchmark": "XSUM", "benchmark_family_key": "helm_classic", "benchmark_family_name": "XSUM", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "XSUM", "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "XSUM", "display_name": "XSUM", "canonical_display_name": "XSUM", "is_summary_score": false, "category": "knowledge", "source_data": { "dataset_name": "XSUM", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "helm_classic", "overview": "helm_classic is a composite benchmark suite designed for holistic evaluation of language models across multiple capabilities and domains. It aggregates several established benchmarks including BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM. The suite measures a wide range of language understanding and generation tasks such as question answering, summarization, reasoning, and bias detection, providing comprehensive coverage of model performance in diverse scenarios.", "benchmark_type": "composite", "contains": [ "BoolQ", "CNN/DailyMail", "CivilComments", "HellaSwag", "IMDB", "MMLU", "MS MARCO (TREC)", "NarrativeQA", "NaturalQuestions (open-book)", "OpenbookQA", "QuAC", "RAFT", "TruthfulQA", "XSUM" ], "data_type": "text", "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "similar_benchmarks": [ "GEM", "XTREME", "GEMv2" ], "resources": [ "https://crfm.stanford.edu/helm/classic/latest/", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json", "https://arxiv.org/abs/2211.09110" ] }, "purpose_and_intended_users": { "goal": "To holistically evaluate language models across a comprehensive set of tasks and domains, including question answering, summarization, reasoning, and classification, by aggregating performance from multiple established benchmarks.", "audience": [ "NLP researchers", "Language model developers" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ], "limitations": "Not specified", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "The benchmark suite aggregates data from multiple existing datasets: BoolQ, CNN/DailyMail, CivilComments, HellaSwag, IMDB, MMLU, MS MARCO (TREC), NarrativeQA, NaturalQuestions (open-book), OpenbookQA, QuAC, RAFT, TruthfulQA, and XSUM.", "size": "Not specified", "format": "Not specified", "annotation": "Not specified" }, "methodology": { "methods": [ "The benchmark evaluates language models across multiple tasks including question answering, summarization, reasoning, and language understanding", "It employs a holistic evaluation framework that aggregates performance from diverse sub-benchmarks" ], "metrics": [ "Exact Match (EM)", "ROUGE-2", "Normalized Discounted Cumulative Gain (NDCG@10)", "F1 score" ], "calculation": "Metrics are calculated per sub-benchmark using standard evaluation protocols: EM for exact answer matching, ROUGE-2 for summarization quality, NDCG@10 for ranking tasks, and F1 for token-level overlap", "interpretation": "Higher scores indicate better performance across all metrics", "baseline_results": "Not specified", "validation": "Not specified" }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Reproducibility", "description": [ "Replicating agent behavior or output can be impacted by changes or updates made to external services and tools. This impact is increased if the agent is built with generative AI." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/reproducibility-agentic.html" } ], "flagged_fields": {}, "missing_fields": [] }, "tags": { "domains": [ "question answering", "summarization", "reasoning", "bias detection", "language understanding", "language generation" ], "languages": [ "English" ], "tasks": [ "Question answering", "Text summarization", "Text classification", "Reasoning", "Toxic comment identification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "ROUGE-2" ], "primary_metric_name": "ROUGE-2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_xsum_rouge_2", "legacy_eval_summary_id": "helm_classic_xsum", "evaluation_name": "XSUM", "display_name": "XSUM / ROUGE-2", "canonical_display_name": "XSUM / ROUGE-2", "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ROUGE-2 on XSUM", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "models_count": 1, "top_score": 0.122, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.122, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "XSUM", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "XSUM", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "xsum", "benchmark_leaf_name": "XSUM", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "XSUM / ROUGE-2", "raw_evaluation_name": "XSUM", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.122, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "language_understanding": [ { "eval_summary_id": "helm_classic_boolq", "benchmark": "BoolQ", "benchmark_family_key": "helm_classic", "benchmark_family_name": "BoolQ", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "BoolQ", "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "BoolQ", "display_name": "BoolQ", "canonical_display_name": "BoolQ", "is_summary_score": false, "category": "language_understanding", "source_data": { "dataset_name": "BoolQ", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "BoolQ", "overview": "BoolQ is a benchmark that measures a model's ability to answer naturally occurring yes/no questions, framed as a reading comprehension task. The questions are generated in unprompted and unconstrained settings, often querying complex, non-factoid information and requiring difficult entailment-like inference. The dataset consists of a single task: answering yes/no questions given a supporting passage.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "natural language understanding", "reading comprehension", "natural language inference" ], "languages": [ "English" ], "similar_benchmarks": [ "MultiNLI", "SNLI", "QNLI", "SQuAD 2.0", "Natural Questions (NQ)", "QQP", "MS MARCO", "RACE", "bAbI stories" ], "resources": [ "https://arxiv.org/abs/1905.10044", "https://huggingface.co/datasets/google/boolq", "https://goo.gl/boolq", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To test models on their ability to answer naturally occurring yes/no questions, which are challenging and require complex inferential abilities beyond surface-level reasoning.", "audience": [ "Researchers in natural language understanding and reading comprehension" ], "tasks": [ "Yes/no question answering", "Text-pair classification" ], "limitations": "Annotation involved some errors and ambiguous cases. The use of singly-annotated examples is a trade-off for dataset size. Potential concerns about annotation artifacts are acknowledged.", "out_of_scope_uses": "The paper does not explicitly state what the benchmark is not designed for." }, "data": { "source": "The data consists of naturally occurring yes/no questions authored by people who were not prompted to write specific question types and did not know the answers. The passages are excerpts from sources like Wikipedia.", "size": "15,942 examples total, with 9,427 in the train split and 3,270 in the validation split. The dataset size category is between 10,000 and 100,000 examples.", "format": "parquet", "annotation": "Questions were answered by human annotators. A quality check on a subset showed the main annotation process achieved 90% accuracy against a gold-standard set labeled by three authors. The training, development, and test sets use singly-annotated examples." }, "methodology": { "methods": [ "Models are evaluated by fine-tuning on the BoolQ training set, potentially after transfer learning from other datasets or unsupervised pre-training. Zero-shot or direct use of pre-trained models without fine-tuning did not outperform the majority baseline.", "The task requires providing a yes/no (boolean) answer to a question based on a given passage." ], "metrics": [ "Accuracy" ], "calculation": "The overall score is the accuracy percentage on the test set.", "interpretation": "Higher accuracy indicates better performance. Human accuracy is 90%, and the majority baseline is approximately 62%.", "baseline_results": "Paper baselines: Majority baseline: 62.17% dev, 62.31% test; Recurrent model baseline: 69.6%; Best model (BERT large pre-trained on MultiNLI then fine-tuned on BoolQ): 80.4% accuracy; Human accuracy: 90%. EEE results: Anthropic-LM v4-s3 52B: 81.5%.", "validation": "Quality assurance involved author-led gold-standard annotation on a subset, showing 90% agreement. The development set was used for model selection, such as choosing the best model from five seeds based on its performance." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "cc-by-sa-3.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:08:51.830946", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "natural language understanding", "reading comprehension", "natural language inference" ], "languages": [ "English" ], "tasks": [ "Yes/no question answering", "Text-pair classification" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_boolq_exact_match", "legacy_eval_summary_id": "helm_classic_boolq", "evaluation_name": "BoolQ", "display_name": "BoolQ / Exact Match", "canonical_display_name": "BoolQ / Exact Match", "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on BoolQ", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.722, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.722, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "BoolQ", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "BoolQ", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "boolq", "benchmark_leaf_name": "BoolQ", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "BoolQ / Exact Match", "raw_evaluation_name": "BoolQ", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.722, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } }, { "eval_summary_id": "helm_classic_cnn_dailymail", "benchmark": "CNN/DailyMail", "benchmark_family_key": "helm_classic", "benchmark_family_name": "CNN/DailyMail", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CNN/DailyMail", "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "CNN/DailyMail", "display_name": "CNN/DailyMail", "canonical_display_name": "CNN/DailyMail", "is_summary_score": false, "category": "language_understanding", "source_data": { "dataset_name": "CNN/DailyMail", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "CNN/DailyMail", "overview": "CNN/DailyMail is a benchmark for evaluating abstractive and extractive summarization models using news articles. It contains over 300,000 unique articles written by journalists from CNN and the Daily Mail. The dataset was originally created for machine reading and question answering, but later versions were restructured specifically for summarization tasks.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "summarization", "journalism", "news media" ], "languages": [ "English" ], "similar_benchmarks": "No facts provided about similar benchmarks.", "resources": [ "https://huggingface.co/datasets/abisee/cnn_dailymail", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To help develop models that can summarize long paragraphs of text into one or two sentences, aiding in the efficient presentation of information from large quantities of text.", "audience": [ "NLP researchers", "Summarization model developers" ], "tasks": [ "Summarization" ], "limitations": "News articles often place important information in the first third, which may affect summarization. A manual study found 25% of samples in an earlier version were difficult for humans due to ambiguity and coreference errors. Also, machine-generated summaries may differ in truth values from the original articles.", "out_of_scope_uses": "No facts provided about out-of-scope uses." }, "data": { "source": "The dataset consists of news articles and highlight sentences written by journalists at CNN and the Daily Mail. The CNN articles were collected from April 2007 to April 2015, and the Daily Mail articles from June 2010 to April 2015, sourced from archives on the Wayback Machine.", "size": "Over 300,000 unique articles, with 287,113 training examples, 13,368 validation examples, and 11,490 test examples.", "format": "parquet", "annotation": "The dataset does not contain additional annotations. The highlights are the original summaries written by the article authors and are used as the target for summarization." }, "methodology": { "methods": [ "Models generate a summary for a given news article, which is then compared to the author-written highlights." ], "metrics": [ "ROUGE-2" ], "calculation": "The ROUGE-2 score measures the overlap of bigrams between the generated summary and the reference highlights.", "interpretation": "Higher scores indicate better performance, as they reflect greater overlap with the reference summaries.", "baseline_results": "Paper baseline (Zhong et al., 2020): ROUGE-1 score of 44.41 for an extractive summarization model. Evaluation suite result (Anthropic-LM v4-s3 52B): ROUGE-2 score of 0.154.", "validation": "No facts provided about validation procedures." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "The dataset (version 3.0.0) is not anonymized, meaning individuals' names are present in the text.", "data_licensing": "Apache License 2.0", "consent_procedures": "Not specified", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Data contamination", "description": [ "Data contamination occurs when incorrect data is used for training. For example, data that is not aligned with model's purpose or data that is already set aside for other development tasks such as testing and evaluation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-contamination.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" } ], "flagged_fields": {}, "missing_fields": [ "ethical_and_legal_considerations.consent_procedures", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:15:47.316103", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "summarization", "journalism", "news media" ], "languages": [ "English" ], "tasks": [ "Summarization" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "ROUGE-2" ], "primary_metric_name": "ROUGE-2", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_cnn_dailymail_rouge_2", "legacy_eval_summary_id": "helm_classic_cnn_dailymail", "evaluation_name": "CNN/DailyMail", "display_name": "CNN/DailyMail / ROUGE-2", "canonical_display_name": "CNN/DailyMail / ROUGE-2", "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "metric_config": { "evaluation_description": "ROUGE-2 on CNN/DailyMail", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "rouge_2", "metric_name": "ROUGE-2", "metric_kind": "rouge", "metric_unit": "proportion", "metric_parameters": { "n": 2 } }, "models_count": 1, "top_score": 0.143, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.143, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CNN/DailyMail", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CNN/DailyMail", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "cnn_dailymail", "benchmark_leaf_name": "CNN/DailyMail", "slice_key": null, "slice_name": null, "metric_name": "ROUGE-2", "metric_id": "rouge_2", "metric_key": "rouge_2", "metric_source": "metric_config", "display_name": "ROUGE-2", "canonical_display_name": "CNN/DailyMail / ROUGE-2", "raw_evaluation_name": "CNN/DailyMail", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.143, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "reasoning": [ { "eval_summary_id": "helm_classic_hellaswag", "benchmark": "HellaSwag", "benchmark_family_key": "helm_classic", "benchmark_family_name": "HellaSwag", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "HellaSwag", "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "HellaSwag", "display_name": "HellaSwag", "canonical_display_name": "HellaSwag", "is_summary_score": false, "category": "reasoning", "source_data": { "dataset_name": "HellaSwag", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "HellaSwag", "overview": "HellaSwag is a benchmark designed to measure commonsense natural language inference by testing a model's ability to select the most plausible follow-up event from four multiple-choice options. It is adversarially constructed to be challenging for state-of-the-art models, using a method called Adversarial Filtering to create difficult wrong answers that are obvious to humans but often misclassified by models.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "text", "domains": [ "commonsense reasoning", "natural language inference" ], "languages": [ "English" ], "similar_benchmarks": [ "SWAG", "SNLI" ], "resources": [ "https://rowanzellers.com/hellaswag", "https://arxiv.org/abs/1905.07830", "https://huggingface.co/datasets/Rowan/hellaswag", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To create a challenge dataset that reveals the difficulty of commonsense inference for state-of-the-art models, demonstrating their lack of robustness and reliance on dataset biases rather than genuine reasoning. It aims to evaluate a model's ability to select the most plausible continuation of a given event description.", "audience": [ "NLP researchers" ], "tasks": [ "Four-way multiple-choice selection for event continuation", "Commonsense inference" ], "limitations": "The adversarial filtering process used to create the dataset, while effective at making it difficult for models, may also select examples where the ground truth answer is not the one preferred by human annotators, necessitating manual filtering to retain the best examples.", "out_of_scope_uses": [ "Not specified" ] }, "data": { "source": "Contexts are sourced from WikiHow instructional articles and ActivityNet video descriptions. Incorrect answer choices are generated by machines and then adversarially filtered.", "size": "The dataset contains 70,000 examples in total, with 5,001 in-domain validation examples and 5,000 zero-shot validation examples. The training set comprises 39,905 examples.", "format": "Parquet", "annotation": "Human crowd workers on Amazon Mechanical Turk validated the endings. They were presented with a context and six endings (one true, five machine-generated) and rated their plausibility. The process involved iterative filtering and replacement of unrealistic endings. Worker quality was ensured via an autograded test and fair pay. A gold standard check by three authors on a random sample showed 90% agreement with crowd annotations." }, "methodology": { "methods": [ "Models are evaluated via fine-tuning on the dataset.", "The benchmark also includes zero-shot evaluation on held-out categories." ], "metrics": [ "HellaSwag accuracy" ], "calculation": "The overall score is the accuracy percentage on the full validation or test sets. Performance is also broken down by subsets, such as in-domain versus zero-shot and by data source.", "interpretation": "Higher scores indicate better performance. Human performance is over 95%, which is considered strong. Model performance below 50% is reported, indicating a struggle, with a gap of over 45% from human performance on in-domain data.", "baseline_results": "Paper baselines: BERT-Large achieves 47.3% accuracy overall. ESIM + ELMo gets 33.3% accuracy. A BERT-Base model with a frozen encoder and an added LSTM performs 4.3% worse than fine-tuned BERT-Base. Evaluation suite results: Anthropic-LM v4-s3 52B achieves 0.807 (80.7%) accuracy.", "validation": "Human validation involved giving five crowd workers the same multiple-choice task and combining their answers via majority vote to establish a human performance baseline. The adversarial filtering process used iterative human ratings to ensure wrong answers were implausible." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "Not specified", "data_licensing": "Not specified", "consent_procedures": "Crowdworkers on Amazon Mechanical Turk participated voluntarily and were compensated, with pay described as fair. A qualification task was used to filter workers, and those who consistently preferred generated endings over real ones were disqualified.", "compliance_with_regulations": "Not specified" }, "possible_risks": [ { "category": "Over- or under-reliance", "description": [ "In AI-assisted decision-making tasks, reliance measures how much a person trusts (and potentially acts on) a model's output. Over-reliance occurs when a person puts too much trust in a model, accepting a model's output when the model's output is likely incorrect. Under-reliance is the opposite, where the person doesn't trust the model but should." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/over-or-under-reliance.html" }, { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Improper usage", "description": [ "Improper usage occurs when a model is used for a purpose that it was not originally designed for." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/improper-usage.html" } ], "flagged_fields": { "baseline_results": "[Possible Hallucination], no supporting evidence found in source material" }, "missing_fields": [ "purpose_and_intended_users.out_of_scope_uses", "ethical_and_legal_considerations.privacy_and_anonymity", "ethical_and_legal_considerations.data_licensing", "ethical_and_legal_considerations.compliance_with_regulations" ], "card_info": { "created_at": "2026-03-17T15:47:07.561060", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "commonsense reasoning", "natural language inference" ], "languages": [ "English" ], "tasks": [ "Four-way multiple-choice selection for event continuation", "Commonsense inference" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_hellaswag_exact_match", "legacy_eval_summary_id": "helm_classic_hellaswag", "evaluation_name": "HellaSwag", "display_name": "HellaSwag / Exact Match", "canonical_display_name": "HellaSwag / Exact Match", "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on HellaSwag", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.739, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.739, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "HellaSwag", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "HellaSwag", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "hellaswag", "benchmark_leaf_name": "HellaSwag", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "HellaSwag / Exact Match", "raw_evaluation_name": "HellaSwag", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.739, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ], "safety": [ { "eval_summary_id": "helm_classic_civilcomments", "benchmark": "CivilComments", "benchmark_family_key": "helm_classic", "benchmark_family_name": "CivilComments", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CivilComments", "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "CivilComments", "display_name": "CivilComments", "canonical_display_name": "CivilComments", "is_summary_score": false, "category": "safety", "source_data": { "dataset_name": "CivilComments", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "benchmark_card": { "benchmark_details": { "name": "CivilComments", "overview": "CivilComments is a benchmark designed to measure unintended identity-based bias in toxicity classification models. It uses a large, real-world dataset of online comments from the Civil Comments platform, extended with crowd-sourced annotations for toxicity and demographic identity references. This provides a nuanced evaluation of bias beyond synthetic datasets.", "benchmark_type": "single", "appears_in": [ "helm_classic" ], "data_type": "tabular, text", "domains": [ "machine learning fairness", "bias measurement", "toxic comment classification", "text classification" ], "languages": [ "English" ], "similar_benchmarks": "The paper does not name other specific benchmark datasets, only referencing prior work using synthetic test sets.", "resources": [ "https://arxiv.org/abs/1903.04561", "https://huggingface.co/datasets/google/civil_comments", "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "purpose_and_intended_users": { "goal": "To evaluate unintended identity-based bias in toxicity classification models using real data and nuanced metrics.", "audience": [ "Machine learning researchers and practitioners working on fairness, bias measurement, and mitigation, particularly in toxic comment classification." ], "tasks": [ "Binary toxicity classification (toxic vs. non-toxic)", "Analysis of performance across identity subgroups" ], "limitations": "The labeled set of identities is not comprehensive and does not provide universal coverage, representing a balance between coverage, annotator accuracy, and example count. The real-world data is potentially noisier than synthetic alternatives.", "out_of_scope_uses": [ "Developing effective strategies for choosing optimal thresholds to minimize bias" ] }, "data": { "source": "The data consists of online comments sourced from the Civil Comments platform, a commenting plugin for independent English-language news sites. The comments were publicly posted between 2015 and 2017.", "size": "The dataset contains approximately 1.8 million comments for training, with separate validation and test sets of approximately 97,320 examples each. All comments were labeled for toxicity, and a subset of 450,000 comments was additionally labeled for identity references.", "format": "parquet", "annotation": "Labeling was performed by crowd raters. Toxicity labels were applied using guidelines consistent with the Perspective API. For the identity-labeled subset, raters were shown comments and selected referenced identities (e.g., genders, races, ethnicities) from a provided list. Some comments for identity labeling were pre-selected by models to increase the frequency of identity content." }, "methodology": { "methods": [ "Models are evaluated by applying a suite of bias metrics to their predictions on the test set. The original paper demonstrates this using publicly accessible toxicity classifiers on the dataset." ], "metrics": [ "Subgroup AUC", "BPSN AUC", "BNSP AUC", "Negative Average Equality Gap (AEG)", "Positive Average Equality Gap (AEG)" ], "calculation": "The evaluation calculates five metrics for each identity subgroup to provide a multi-faceted view of bias. There is no single aggregated overall score.", "interpretation": "For the AUC metrics (Subgroup, BPSN, BNSP), higher values indicate better separability (fewer mis-orderings). For the Average Equality Gaps (Negative and Positive), lower values indicate better separability (more similar score distributions).", "baseline_results": "Paper baselines: Results for TOXICITY@1 and TOXICITY@6 from the Perspective API are reported, showing their Subgroup AUC, BPSN AUC, BNSP AUC, Negative AEG, and Positive AEG on a synthetic dataset for the lowest performing 20 subgroups. They are also compared on short comments within the human-labeled dataset for specific identities. EEE results: Anthropic-LM v4-s3 52B scored 0.6100 on the CivilComments metric.", "validation": "The evaluation assumes the human-provided labels are reliable. The identity labeling set was designed to balance coverage, crowd rater accuracy, and ensure sufficient examples per identity for meaningful results." }, "ethical_and_legal_considerations": { "privacy_and_anonymity": "The paper does not discuss how personally identifiable information (PII) in the online comments was handled or if data was anonymized.", "data_licensing": "Creative Commons Zero v1.0 Universal", "consent_procedures": "The paper does not describe compensation for crowdworkers or the specific platform used for annotation.", "compliance_with_regulations": "The paper does not mention IRB approval, GDPR compliance, or any other ethical review process." }, "possible_risks": [ { "category": "Unrepresentative data", "description": [ "Unrepresentative data occurs when the training or fine-tuning data is not sufficiently representative of the underlying population or does not measure the phenomenon of interest. Synthetic data might not fully capture the complexity and nuances of real-world data. Causes include possible limitations in the seed data quality, biases in generation methods, or inadequate domain knowledge. Thus, AI models might struggle to generalize effectively to real-world scenarios." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/unrepresentative-data.html" }, { "category": "Uncertain data provenance", "description": [ "Data provenance refers to the traceability of data (including synthetic data), which includes its ownership, origin, transformations, and generation. Proving that the data is the same as the original source with correct usage terms is difficult without standardized methods for verifying data sources or generation." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-provenance.html" }, { "category": "Data bias", "description": [ "Historical and societal biases might be present in data that are used to train and fine-tune models. Biases can also be inherited from seed data or exacerbated by synthetic data generation methods." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/data-bias.html" }, { "category": "Lack of data transparency", "description": [ "Lack of data transparency might be due to insufficient documentation of training or tuning dataset details, including synthetic data generation. " ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/lack-of-data-transparency.html" }, { "category": "Output bias", "description": [ "Generated content might unfairly represent certain groups or individuals." ], "url": "https://www.ibm.com/docs/en/watsonx/saas?topic=SSYOK8/wsj/ai-risk-atlas/output-bias.html" } ], "flagged_fields": {}, "missing_fields": [], "card_info": { "created_at": "2026-03-17T12:38:43.250822", "llm": "deepseek-ai/DeepSeek-V3.2" } }, "tags": { "domains": [ "machine learning fairness", "bias measurement", "toxic comment classification", "text classification" ], "languages": [ "English" ], "tasks": [ "Binary toxicity classification (toxic vs. non-toxic)", "Analysis of performance across identity subgroups" ] }, "subtasks_count": 0, "metrics_count": 1, "metric_names": [ "Exact Match" ], "primary_metric_name": "Exact Match", "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.9285714285714286, "total_fields_evaluated": 28, "missing_required_fields": [ "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 1.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 1, "has_reproducibility_gap_count": 1, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 1, "total_groups": 1, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 1, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 1, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 }, "metrics": [ { "metric_summary_id": "helm_classic_civilcomments_exact_match", "legacy_eval_summary_id": "helm_classic_civilcomments", "evaluation_name": "CivilComments", "display_name": "CivilComments / Exact Match", "canonical_display_name": "CivilComments / Exact Match", "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "metric_config": { "evaluation_description": "EM on CivilComments", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0, "metric_id": "exact_match", "metric_name": "Exact Match", "metric_kind": "exact_match", "metric_unit": "proportion" }, "models_count": 1, "top_score": 0.529, "model_results": [ { "model_id": "ai21/j1-grande-v1-17b", "model_route_id": "ai21__j1-grande-v1-17b", "model_name": "J1-Grande v1 17B", "developer": "ai21", "variant_key": "default", "raw_model_id": "ai21/J1-Grande-v1-17B", "score": 0.529, "evaluation_id": "helm_classic/ai21_J1-Grande-v1-17B/1774096308.339228", "retrieved_timestamp": "1774096308.339228", "source_metadata": { "source_name": "helm_classic", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_classic", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/benchmark_output/releases/v0.4.0/groups/core_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ai21__j1-grande-v1-17b/helm_classic_ai21_j1_grande_v1_17b_1774096308_339228.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_classic", "benchmark_family_name": "CivilComments", "benchmark_parent_key": "helm_classic", "benchmark_parent_name": "CivilComments", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "civilcomments", "benchmark_leaf_name": "CivilComments", "slice_key": null, "slice_name": null, "metric_name": "Exact Match", "metric_id": "exact_match", "metric_key": "exact_match", "metric_source": "metric_config", "display_name": "Exact Match", "canonical_display_name": "CivilComments / Exact Match", "raw_evaluation_name": "CivilComments", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ] } ], "subtasks": [], "models_count": 1, "top_score": 0.529, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 } } ] }, "total_evaluations": 1, "last_updated": "2026-03-21T12:31:48.339228Z", "categories_covered": [ "knowledge", "language_understanding", "reasoning", "safety" ], "variants": [ { "variant_key": "default", "variant_label": "Default", "evaluation_count": 1, "raw_model_ids": [ "ai21/J1-Grande-v1-17B" ], "last_updated": "2026-03-21T12:31:48.339228Z" } ], "reproducibility_summary": { "results_total": 15, "has_reproducibility_gap_count": 15, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 15, "total_groups": 15, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 15, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 15, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 } }