{ "eval_summary_id": "helm_safety_simplesafetytests", "benchmark": "SimpleSafetyTests", "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "evaluation_name": "SimpleSafetyTests", "display_name": "SimpleSafetyTests", "canonical_display_name": "SimpleSafetyTests", "is_summary_score": false, "category": "general", "source_data": { "dataset_name": "SimpleSafetyTests", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "benchmark_card": null, "tags": { "domains": [], "languages": [], "tasks": [] }, "subtasks": [], "metrics": [ { "metric_summary_id": "helm_safety_simplesafetytests_lm_evaluated_safety_score", "legacy_eval_summary_id": "helm_safety_simplesafetytests", "evaluation_name": "SimpleSafetyTests", "display_name": "SimpleSafetyTests / LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "lower_is_better": false, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "metric_config": { "evaluation_description": "LM Evaluated Safety score on SimpleSafetyTests", "lower_is_better": false, "score_type": "continuous", "min_score": 0.0, "max_score": 1.0 }, "model_results": [ { "model_id": "writer/palmyra-x5", "model_route_id": "writer__palmyra-x5", "model_name": "Palmyra X5", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-x5", "score": 1.0, "evaluation_id": "helm_safety/writer_palmyra-x5/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x5/helm_safety_writer_palmyra_x5_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-x-004", "model_route_id": "writer__palmyra-x-004", "model_name": "Palmyra-X-004", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-x-004", "score": 1.0, "evaluation_id": "helm_safety/writer_palmyra-x-004/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-004/helm_safety_writer_palmyra_x_004_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-fin", "model_route_id": "writer__palmyra-fin", "model_name": "Palmyra Fin", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-fin", "score": 1.0, "evaluation_id": "helm_safety/writer_palmyra-fin/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-fin/helm_safety_writer_palmyra_fin_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8", "model_route_id": "qwen__qwen3-235b-a22b-instruct-2507-fp8", "model_name": "Qwen3 235B A22B Instruct 2507 FP8", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8", "score": 1.0, "evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-instruct-2507-fp8/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-instruct-2507-fp8/helm_safety_qwen_qwen3_235b_a22b_instruct_2507_fp8_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-5-72b-instruct-turbo", "model_route_id": "qwen__qwen2-5-72b-instruct-turbo", "model_name": "Qwen2.5 Instruct Turbo 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2.5-72b-instruct-turbo", "score": 1.0, "evaluation_id": "helm_safety/qwen_qwen2.5-72b-instruct-turbo/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-72b-instruct-turbo/helm_safety_qwen_qwen2_5_72b_instruct_turbo_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/o4-mini", "model_route_id": "openai__o4-mini", "model_name": "o4-mini 2025-04-16", "developer": "openai", "variant_key": "2025-04-16", "raw_model_id": "openai/o4-mini-2025-04-16", "score": 1.0, "evaluation_id": "helm_safety/openai_o4-mini-2025-04-16/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o4-mini/helm_safety_openai_o4_mini_2025_04_16_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-oss-20b", "model_route_id": "openai__gpt-oss-20b", "model_name": "gpt-oss-20b", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-oss-20b", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-oss-20b/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-20b/helm_safety_openai_gpt_oss_20b_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-oss-120b", "model_route_id": "openai__gpt-oss-120b", "model_name": "gpt-oss-120b", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-oss-120b", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-oss-120b/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-120b/helm_safety_openai_gpt_oss_120b_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-nano", "model_route_id": "openai__gpt-5-nano", "model_name": "GPT-5 nano 2025-08-07", "developer": "openai", "variant_key": "2025-08-07", "raw_model_id": "openai/gpt-5-nano-2025-08-07", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-5-nano-2025-08-07/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-nano/helm_safety_openai_gpt_5_nano_2025_08_07_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-mini", "model_route_id": "openai__gpt-5-mini", "model_name": "GPT-5 mini 2025-08-07", "developer": "openai", "variant_key": "2025-08-07", "raw_model_id": "openai/gpt-5-mini-2025-08-07", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-5-mini-2025-08-07/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-mini/helm_safety_openai_gpt_5_mini_2025_08_07_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-5-preview", "model_route_id": "openai__gpt-4-5-preview", "model_name": "GPT-4.5 2025-02-27 preview", "developer": "openai", "variant_key": "2025-02-27", "raw_model_id": "openai/gpt-4.5-preview-2025-02-27", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-4.5-preview-2025-02-27/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-5-preview/helm_safety_openai_gpt_4_5_preview_2025_02_27_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-1-mini", "model_route_id": "openai__gpt-4-1-mini", "model_name": "GPT-4.1 mini 2025-04-14", "developer": "openai", "variant_key": "2025-04-14", "raw_model_id": "openai/gpt-4.1-mini-2025-04-14", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-4.1-mini-2025-04-14/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-mini/helm_safety_openai_gpt_4_1_mini_2025_04_14_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-1", "model_route_id": "openai__gpt-4-1", "model_name": "GPT-4.1 2025-04-14", "developer": "openai", "variant_key": "2025-04-14", "raw_model_id": "openai/gpt-4.1-2025-04-14", "score": 1.0, "evaluation_id": "helm_safety/openai_gpt-4.1-2025-04-14/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1/helm_safety_openai_gpt_4_1_2025_04_14_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "moonshotai/kimi-k2-instruct", "model_route_id": "moonshotai__kimi-k2-instruct", "model_name": "Kimi K2 Instruct", "developer": "moonshotai", "variant_key": "default", "raw_model_id": "moonshotai/kimi-k2-instruct", "score": 1.0, "evaluation_id": "helm_safety/moonshotai_kimi-k2-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/moonshotai__kimi-k2-instruct/helm_safety_moonshotai_kimi_k2_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ibm/granite-4-0-micro-with-guardian", "model_route_id": "ibm__granite-4-0-micro-with-guardian", "model_name": "IBM Granite 4.0 Micro with guardian", "developer": "ibm", "variant_key": "default", "raw_model_id": "ibm/granite-4.0-micro-with-guardian", "score": 1.0, "evaluation_id": "helm_safety/ibm_granite-4.0-micro-with-guardian/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro-with-guardian/helm_safety_ibm_granite_4_0_micro_with_guardian_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ibm/granite-4-0-h-small-with-guardian", "model_route_id": "ibm__granite-4-0-h-small-with-guardian", "model_name": "IBM Granite 4.0 Small with guardian", "developer": "ibm", "variant_key": "default", "raw_model_id": "ibm/granite-4.0-h-small-with-guardian", "score": 1.0, "evaluation_id": "helm_safety/ibm_granite-4.0-h-small-with-guardian/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small-with-guardian/helm_safety_ibm_granite_4_0_h_small_with_guardian_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command-r-plus", "model_route_id": "cohere__command-r-plus", "model_name": "Command R Plus", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command-r-plus", "score": 1.0, "evaluation_id": "helm_safety/cohere_command-r-plus/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r-plus/helm_safety_cohere_command_r_plus_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-sonnet-4-5", "model_route_id": "anthropic__claude-sonnet-4-5", "model_name": "Claude 4.5 Sonnet 20250929", "developer": "anthropic", "variant_key": "20250929", "raw_model_id": "anthropic/claude-sonnet-4-5-20250929", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-5-20250929/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4-5/helm_safety_anthropic_claude_sonnet_4_5_20250929_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-sonnet-4", "model_route_id": "anthropic__claude-sonnet-4", "model_name": "Claude 4 Sonnet 20250514, extended thinking", "developer": "anthropic", "variant_key": "20250514-thinking-10k", "raw_model_id": "anthropic/claude-sonnet-4-20250514-thinking-10k", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514-thinking-10k/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_thinking_10k_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-opus-4", "model_route_id": "anthropic__claude-opus-4", "model_name": "Claude 4 Opus 20250514, extended thinking", "developer": "anthropic", "variant_key": "20250514-thinking-10k", "raw_model_id": "anthropic/claude-opus-4-20250514-thinking-10k", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514-thinking-10k/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_thinking_10k_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-sonnet", "model_route_id": "anthropic__claude-3-sonnet", "model_name": "Claude 3 Sonnet 20240229", "developer": "anthropic", "variant_key": "20240229", "raw_model_id": "anthropic/claude-3-sonnet-20240229", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-3-sonnet-20240229/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-sonnet/helm_safety_anthropic_claude_3_sonnet_20240229_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-opus", "model_route_id": "anthropic__claude-3-opus", "model_name": "Claude 3 Opus 20240229", "developer": "anthropic", "variant_key": "20240229", "raw_model_id": "anthropic/claude-3-opus-20240229", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-3-opus-20240229/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-opus/helm_safety_anthropic_claude_3_opus_20240229_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-haiku", "model_route_id": "anthropic__claude-3-haiku", "model_name": "Claude 3 Haiku 20240307", "developer": "anthropic", "variant_key": "20240307", "raw_model_id": "anthropic/claude-3-haiku-20240307", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-3-haiku-20240307/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-haiku/helm_safety_anthropic_claude_3_haiku_20240307_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-7-sonnet", "model_route_id": "anthropic__claude-3-7-sonnet", "model_name": "Claude 3.7 Sonnet 20250219", "developer": "anthropic", "variant_key": "20250219", "raw_model_id": "anthropic/claude-3-7-sonnet-20250219", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-3-7-sonnet-20250219/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-7-sonnet/helm_safety_anthropic_claude_3_7_sonnet_20250219_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-3-5-sonnet", "model_route_id": "anthropic__claude-3-5-sonnet", "model_name": "Claude 3.5 Sonnet 20240620", "developer": "anthropic", "variant_key": "20240620", "raw_model_id": "anthropic/claude-3-5-sonnet-20240620", "score": 1.0, "evaluation_id": "helm_safety/anthropic_claude-3-5-sonnet-20240620/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-sonnet/helm_safety_anthropic_claude_3_5_sonnet_20240620_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5-1", "model_route_id": "openai__gpt-5-1", "model_name": "GPT-5.1 2025-11-13", "developer": "openai", "variant_key": "2025-11-13", "raw_model_id": "openai/gpt-5.1-2025-11-13", "score": 0.998, "evaluation_id": "helm_safety/openai_gpt-5.1-2025-11-13/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-1/helm_safety_openai_gpt_5_1_2025_11_13_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-5", "model_route_id": "openai__gpt-5", "model_name": "GPT-5 2025-08-07", "developer": "openai", "variant_key": "2025-08-07", "raw_model_id": "openai/gpt-5-2025-08-07", "score": 0.998, "evaluation_id": "helm_safety/openai_gpt-5-2025-08-07/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5/helm_safety_openai_gpt_5_2025_08_07_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen3-next-80b-a3b-thinking", "model_route_id": "qwen__qwen3-next-80b-a3b-thinking", "model_name": "Qwen3-Next 80B A3B Thinking", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen3-next-80b-a3b-thinking", "score": 0.995, "evaluation_id": "helm_safety/qwen_qwen3-next-80b-a3b-thinking/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-next-80b-a3b-thinking/helm_safety_qwen_qwen3_next_80b_a3b_thinking_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-sonnet-4", "model_route_id": "anthropic__claude-sonnet-4", "model_name": "Claude 4 Sonnet 20250514", "developer": "anthropic", "variant_key": "20250514", "raw_model_id": "anthropic/claude-sonnet-4-20250514", "score": 0.995, "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-opus-4", "model_route_id": "anthropic__claude-opus-4", "model_name": "Claude 4 Opus 20250514", "developer": "anthropic", "variant_key": "20250514", "raw_model_id": "anthropic/claude-opus-4-20250514", "score": 0.995, "evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "xai/grok-3-mini-beta", "model_route_id": "xai__grok-3-mini-beta", "model_name": "Grok 3 mini Beta", "developer": "xai", "variant_key": "default", "raw_model_id": "xai/grok-3-mini-beta", "score": 0.993, "evaluation_id": "helm_safety/xai_grok-3-mini-beta/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-mini-beta/helm_safety_xai_grok_3_mini_beta_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8", "model_route_id": "meta__llama-4-maverick-17b-128e-instruct-fp8", "model_name": "Llama 4 Maverick 17Bx128E Instruct FP8", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8", "score": 0.993, "evaluation_id": "helm_safety/meta_llama-4-maverick-17b-128e-instruct-fp8/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-maverick-17b-128e-instruct-fp8/helm_safety_meta_llama_4_maverick_17b_128e_instruct_fp8_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-8b-chat", "model_route_id": "meta__llama-3-8b-chat", "model_name": "Llama 3 Instruct 8B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3-8b-chat", "score": 0.993, "evaluation_id": "helm_safety/meta_llama-3-8b-chat/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-8b-chat/helm_safety_meta_llama_3_8b_chat_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "zai-org/glm-4-5-air-fp8", "model_route_id": "zai-org__glm-4-5-air-fp8", "model_name": "GLM-4.5-Air-FP8", "developer": "zai-org", "variant_key": "default", "raw_model_id": "zai-org/glm-4.5-air-fp8", "score": 0.99, "evaluation_id": "helm_safety/zai-org_glm-4.5-air-fp8/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/zai-org__glm-4-5-air-fp8/helm_safety_zai_org_glm_4_5_air_fp8_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen1-5-72b-chat", "model_route_id": "qwen__qwen1-5-72b-chat", "model_name": "Qwen1.5 Chat 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen1.5-72b-chat", "score": 0.99, "evaluation_id": "helm_safety/qwen_qwen1.5-72b-chat/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-72b-chat/helm_safety_qwen_qwen1_5_72b_chat_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/o3-mini", "model_route_id": "openai__o3-mini", "model_name": "o3-mini 2025-01-31", "developer": "openai", "variant_key": "2025-01-31", "raw_model_id": "openai/o3-mini-2025-01-31", "score": 0.99, "evaluation_id": "helm_safety/openai_o3-mini-2025-01-31/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3-mini/helm_safety_openai_o3_mini_2025_01_31_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/o3", "model_route_id": "openai__o3", "model_name": "o3 2025-04-16", "developer": "openai", "variant_key": "2025-04-16", "raw_model_id": "openai/o3-2025-04-16", "score": 0.99, "evaluation_id": "helm_safety/openai_o3-2025-04-16/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3/helm_safety_openai_o3_2025_04_16_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/o1", "model_route_id": "openai__o1", "model_name": "o1 2024-12-17", "developer": "openai", "variant_key": "2024-12-17", "raw_model_id": "openai/o1-2024-12-17", "score": 0.99, "evaluation_id": "helm_safety/openai_o1-2024-12-17/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1/helm_safety_openai_o1_2024_12_17_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-turbo", "model_route_id": "openai__gpt-4-turbo", "model_name": "GPT-4 Turbo 2024-04-09", "developer": "openai", "variant_key": "2024-04-09", "raw_model_id": "openai/gpt-4-turbo-2024-04-09", "score": 0.99, "evaluation_id": "helm_safety/openai_gpt-4-turbo-2024-04-09/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-turbo/helm_safety_openai_gpt_4_turbo_2024_04_09_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4-1-nano", "model_route_id": "openai__gpt-4-1-nano", "model_name": "GPT-4.1 nano 2025-04-14", "developer": "openai", "variant_key": "2025-04-14", "raw_model_id": "openai/gpt-4.1-nano-2025-04-14", "score": 0.99, "evaluation_id": "helm_safety/openai_gpt-4.1-nano-2025-04-14/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-nano/helm_safety_openai_gpt_4_1_nano_2025_04_14_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-70b-chat", "model_route_id": "meta__llama-3-70b-chat", "model_name": "Llama 3 Instruct 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3-70b-chat", "score": 0.99, "evaluation_id": "helm_safety/meta_llama-3-70b-chat/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-70b-chat/helm_safety_meta_llama_3_70b_chat_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-8b-instruct-turbo", "model_route_id": "meta__llama-3-1-8b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 8B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-8b-instruct-turbo", "score": 0.988, "evaluation_id": "helm_safety/meta_llama-3.1-8b-instruct-turbo/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-8b-instruct-turbo/helm_safety_meta_llama_3_1_8b_instruct_turbo_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-405b-instruct-turbo", "model_route_id": "meta__llama-3-1-405b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 405B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-405b-instruct-turbo", "score": 0.988, "evaluation_id": "helm_safety/meta_llama-3.1-405b-instruct-turbo/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-405b-instruct-turbo/helm_safety_meta_llama_3_1_405b_instruct_turbo_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "anthropic/claude-haiku-4-5", "model_route_id": "anthropic__claude-haiku-4-5", "model_name": "Claude 4.5 Haiku 20251001", "developer": "anthropic", "variant_key": "20251001", "raw_model_id": "anthropic/claude-haiku-4-5-20251001", "score": 0.988, "evaluation_id": "helm_safety/anthropic_claude-haiku-4-5-20251001/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-haiku-4-5/helm_safety_anthropic_claude_haiku_4_5_20251001_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "writer/palmyra-med", "model_route_id": "writer__palmyra-med", "model_name": "Palmyra Med", "developer": "writer", "variant_key": "default", "raw_model_id": "writer/palmyra-med", "score": 0.985, "evaluation_id": "helm_safety/writer_palmyra-med/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-med/helm_safety_writer_palmyra_med_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen3-235b-a22b-fp8-tput", "model_route_id": "qwen__qwen3-235b-a22b-fp8-tput", "model_name": "Qwen3 235B A22B FP8 Throughput", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen3-235b-a22b-fp8-tput", "score": 0.985, "evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-fp8-tput/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-fp8-tput/helm_safety_qwen_qwen3_235b_a22b_fp8_tput_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-72b-instruct", "model_route_id": "qwen__qwen2-72b-instruct", "model_name": "Qwen2 Instruct 72B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2-72b-instruct", "score": 0.985, "evaluation_id": "helm_safety/qwen_qwen2-72b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-72b-instruct/helm_safety_qwen_qwen2_72b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4o", "model_route_id": "openai__gpt-4o", "model_name": "GPT-4o 2024-05-13", "developer": "openai", "variant_key": "2024-05-13", "raw_model_id": "openai/gpt-4o-2024-05-13", "score": 0.985, "evaluation_id": "helm_safety/openai_gpt-4o-2024-05-13/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o/helm_safety_openai_gpt_4o_2024_05_13_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-0-flash-001", "model_route_id": "google__gemini-2-0-flash-001", "model_name": "Gemini 2.0 Flash", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.0-flash-001", "score": 0.985, "evaluation_id": "helm_safety/google_gemini-2.0-flash-001/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-001/helm_safety_google_gemini_2_0_flash_001_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-r1-0528", "model_route_id": "deepseek-ai__deepseek-r1-0528", "model_name": "DeepSeek-R1-0528", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-r1-0528", "score": 0.983, "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-0528/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-0528/helm_safety_deepseek_ai_deepseek_r1_0528_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ibm/granite-4-0-h-small", "model_route_id": "ibm__granite-4-0-h-small", "model_name": "IBM Granite 4.0 Small", "developer": "ibm", "variant_key": "default", "raw_model_id": "ibm/granite-4.0-h-small", "score": 0.98, "evaluation_id": "helm_safety/ibm_granite-4.0-h-small/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small/helm_safety_ibm_granite_4_0_h_small_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-5-flash-preview-04-17", "model_route_id": "google__gemini-2-5-flash-preview-04-17", "model_name": "Gemini 2.5 Flash 04-17 preview", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.5-flash-preview-04-17", "score": 0.98, "evaluation_id": "helm_safety/google_gemini-2.5-flash-preview-04-17/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-preview-04-17/helm_safety_google_gemini_2_5_flash_preview_04_17_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-r1-hide-reasoning", "model_route_id": "deepseek-ai__deepseek-r1-hide-reasoning", "model_name": "DeepSeek R1 hide reasoning", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-r1-hide-reasoning", "score": 0.98, "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-hide-reasoning/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-hide-reasoning/helm_safety_deepseek_ai_deepseek_r1_hide_reasoning_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "allenai/olmo-2-0325-32b-instruct", "model_route_id": "allenai__olmo-2-0325-32b-instruct", "model_name": "OLMo 2 32B Instruct March 2025", "developer": "allenai", "variant_key": "default", "raw_model_id": "allenai/olmo-2-0325-32b-instruct", "score": 0.98, "evaluation_id": "helm_safety/allenai_olmo-2-0325-32b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-0325-32b-instruct/helm_safety_allenai_olmo_2_0325_32b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-4o-mini", "model_route_id": "openai__gpt-4o-mini", "model_name": "GPT-4o mini 2024-07-18", "developer": "openai", "variant_key": "2024-07-18", "raw_model_id": "openai/gpt-4o-mini-2024-07-18", "score": 0.978, "evaluation_id": "helm_safety/openai_gpt-4o-mini-2024-07-18/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o-mini/helm_safety_openai_gpt_4o_mini_2024_07_18_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-0-flash-lite-preview-02-05", "model_route_id": "google__gemini-2-0-flash-lite-preview-02-05", "model_name": "Gemini 2.0 Flash Lite 02-05 preview", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.0-flash-lite-preview-02-05", "score": 0.977, "evaluation_id": "helm_safety/google_gemini-2.0-flash-lite-preview-02-05/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-lite-preview-02-05/helm_safety_google_gemini_2_0_flash_lite_preview_02_05_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ibm/granite-3-3-8b-instruct", "model_route_id": "ibm__granite-3-3-8b-instruct", "model_name": "IBM Granite 3.3 8B Instruct", "developer": "ibm", "variant_key": "default", "raw_model_id": "ibm/granite-3.3-8b-instruct", "score": 0.975, "evaluation_id": "helm_safety/ibm_granite-3.3-8b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-3-3-8b-instruct/helm_safety_ibm_granite_3_3_8b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-3-pro-preview", "model_route_id": "google__gemini-3-pro-preview", "model_name": "Gemini 3 Pro Preview", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-3-pro-preview", "score": 0.975, "evaluation_id": "helm_safety/google_gemini-3-pro-preview/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-pro-preview/helm_safety_google_gemini_3_pro_preview_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-0-pro-exp-02-05", "model_route_id": "google__gemini-2-0-pro-exp-02-05", "model_name": "Gemini 2.0 Pro 02-05 preview", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.0-pro-exp-02-05", "score": 0.975, "evaluation_id": "helm_safety/google_gemini-2.0-pro-exp-02-05/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-pro-exp-02-05/helm_safety_google_gemini_2_0_pro_exp_02_05_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-pro-001", "model_route_id": "google__gemini-1-5-pro-001", "model_name": "Gemini 1.5 Pro 001", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-pro-001", "score": 0.975, "evaluation_id": "helm_safety/google_gemini-1.5-pro-001/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-pro-001/helm_safety_google_gemini_1_5_pro_001_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-r1", "model_route_id": "deepseek-ai__deepseek-r1", "model_name": "DeepSeek R1", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-r1", "score": 0.975, "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1/helm_safety_deepseek_ai_deepseek_r1_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/o1-mini", "model_route_id": "openai__o1-mini", "model_name": "o1-mini 2024-09-12", "developer": "openai", "variant_key": "2024-09-12", "raw_model_id": "openai/o1-mini-2024-09-12", "score": 0.97, "evaluation_id": "helm_safety/openai_o1-mini-2024-09-12/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1-mini/helm_safety_openai_o1_mini_2024_09_12_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-4-scout-17b-16e-instruct", "model_route_id": "meta__llama-4-scout-17b-16e-instruct", "model_name": "Llama 4 Scout 17Bx16E Instruct", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-4-scout-17b-16e-instruct", "score": 0.97, "evaluation_id": "helm_safety/meta_llama-4-scout-17b-16e-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-scout-17b-16e-instruct/helm_safety_meta_llama_4_scout_17b_16e_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-5-pro-preview-03-25", "model_route_id": "google__gemini-2-5-pro-preview-03-25", "model_name": "Gemini 2.5 Pro 03-25 preview", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.5-pro-preview-03-25", "score": 0.97, "evaluation_id": "helm_safety/google_gemini-2.5-pro-preview-03-25/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-pro-preview-03-25/helm_safety_google_gemini_2_5_pro_preview_03_25_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-1-5-flash-001", "model_route_id": "google__gemini-1-5-flash-001", "model_name": "Gemini 1.5 Flash 001", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-1.5-flash-001", "score": 0.97, "evaluation_id": "helm_safety/google_gemini-1.5-flash-001/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-flash-001/helm_safety_google_gemini_1_5_flash_001_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "xai/grok-3-beta", "model_route_id": "xai__grok-3-beta", "model_name": "Grok 3 Beta", "developer": "xai", "variant_key": "default", "raw_model_id": "xai/grok-3-beta", "score": 0.968, "evaluation_id": "helm_safety/xai_grok-3-beta/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-beta/helm_safety_xai_grok_3_beta_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-llm-67b-chat", "model_route_id": "deepseek-ai__deepseek-llm-67b-chat", "model_name": "DeepSeek LLM Chat 67B", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-llm-67b-chat", "score": 0.968, "evaluation_id": "helm_safety/deepseek-ai_deepseek-llm-67b-chat/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-llm-67b-chat/helm_safety_deepseek_ai_deepseek_llm_67b_chat_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "google/gemini-2-5-flash-lite", "model_route_id": "google__gemini-2-5-flash-lite", "model_name": "Gemini 2.5 Flash-Lite", "developer": "google", "variant_key": "default", "raw_model_id": "google/gemini-2.5-flash-lite", "score": 0.965, "evaluation_id": "helm_safety/google_gemini-2.5-flash-lite/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-lite/helm_safety_google_gemini_2_5_flash_lite_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "qwen/qwen2-5-7b-instruct-turbo", "model_route_id": "qwen__qwen2-5-7b-instruct-turbo", "model_name": "Qwen2.5 Instruct Turbo 7B", "developer": "qwen", "variant_key": "default", "raw_model_id": "qwen/qwen2.5-7b-instruct-turbo", "score": 0.96, "evaluation_id": "helm_safety/qwen_qwen2.5-7b-instruct-turbo/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-7b-instruct-turbo/helm_safety_qwen_qwen2_5_7b_instruct_turbo_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-3-5-turbo-0613", "model_route_id": "openai__gpt-3-5-turbo-0613", "model_name": "GPT-3.5 Turbo 0613", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-3.5-turbo-0613", "score": 0.958, "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0613/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0613/helm_safety_openai_gpt_3_5_turbo_0613_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "marin-community/marin-8b-instruct", "model_route_id": "marin-community__marin-8b-instruct", "model_name": "Marin 8B Instruct", "developer": "marin-community", "variant_key": "default", "raw_model_id": "marin-community/marin-8b-instruct", "score": 0.958, "evaluation_id": "helm_safety/marin-community_marin-8b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/marin-community__marin-8b-instruct/helm_safety_marin_community_marin_8b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "deepseek-ai/deepseek-v3", "model_route_id": "deepseek-ai__deepseek-v3", "model_name": "DeepSeek v3", "developer": "deepseek-ai", "variant_key": "default", "raw_model_id": "deepseek-ai/deepseek-v3", "score": 0.953, "evaluation_id": "helm_safety/deepseek-ai_deepseek-v3/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-v3/helm_safety_deepseek_ai_deepseek_v3_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "ibm/granite-4-0-micro", "model_route_id": "ibm__granite-4-0-micro", "model_name": "IBM Granite 4.0 Micro", "developer": "ibm", "variant_key": "default", "raw_model_id": "ibm/granite-4.0-micro", "score": 0.945, "evaluation_id": "helm_safety/ibm_granite-4.0-micro/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro/helm_safety_ibm_granite_4_0_micro_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "cohere/command-r", "model_route_id": "cohere__command-r", "model_name": "Command R", "developer": "cohere", "variant_key": "default", "raw_model_id": "cohere/command-r", "score": 0.943, "evaluation_id": "helm_safety/cohere_command-r/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r/helm_safety_cohere_command_r_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-small-2501", "model_route_id": "mistralai__mistral-small-2501", "model_name": "Mistral Small 3 2501", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-small-2501", "score": 0.932, "evaluation_id": "helm_safety/mistralai_mistral-small-2501/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-small-2501/helm_safety_mistralai_mistral_small_2501_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "meta/llama-3-1-70b-instruct-turbo", "model_route_id": "meta__llama-3-1-70b-instruct-turbo", "model_name": "Llama 3.1 Instruct Turbo 70B", "developer": "meta", "variant_key": "default", "raw_model_id": "meta/llama-3.1-70b-instruct-turbo", "score": 0.925, "evaluation_id": "helm_safety/meta_llama-3.1-70b-instruct-turbo/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-70b-instruct-turbo/helm_safety_meta_llama_3_1_70b_instruct_turbo_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "xai/grok-4-0709", "model_route_id": "xai__grok-4-0709", "model_name": "Grok 4 0709", "developer": "xai", "variant_key": "default", "raw_model_id": "xai/grok-4-0709", "score": 0.922, "evaluation_id": "helm_safety/xai_grok-4-0709/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-4-0709/helm_safety_xai_grok_4_0709_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-3-5-turbo-1106", "model_route_id": "openai__gpt-3-5-turbo-1106", "model_name": "GPT-3.5 Turbo 1106", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-3.5-turbo-1106", "score": 0.922, "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-1106/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-1106/helm_safety_openai_gpt_3_5_turbo_1106_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "openai/gpt-3-5-turbo-0125", "model_route_id": "openai__gpt-3-5-turbo-0125", "model_name": "GPT-3.5 Turbo 0125", "developer": "openai", "variant_key": "default", "raw_model_id": "openai/gpt-3.5-turbo-0125", "score": 0.92, "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0125/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0125/helm_safety_openai_gpt_3_5_turbo_0125_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mixtral-8x22b-instruct-v0-1", "model_route_id": "mistralai__mixtral-8x22b-instruct-v0-1", "model_name": "Mixtral Instruct 8x22B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mixtral-8x22b-instruct-v0.1", "score": 0.915, "evaluation_id": "helm_safety/mistralai_mixtral-8x22b-instruct-v0.1/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x22b-instruct-v0-1/helm_safety_mistralai_mixtral_8x22b_instruct_v0_1_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mixtral-8x7b-instruct-v0-1", "model_route_id": "mistralai__mixtral-8x7b-instruct-v0-1", "model_name": "Mixtral Instruct 8x7B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mixtral-8x7b-instruct-v0.1", "score": 0.905, "evaluation_id": "helm_safety/mistralai_mixtral-8x7b-instruct-v0.1/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x7b-instruct-v0-1/helm_safety_mistralai_mixtral_8x7b_instruct_v0_1_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-7b-instruct-v0-3", "model_route_id": "mistralai__mistral-7b-instruct-v0-3", "model_name": "Mistral Instruct v0.3 7B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-7b-instruct-v0.3", "score": 0.81, "evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.3/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-3/helm_safety_mistralai_mistral_7b_instruct_v0_3_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "allenai/olmo-2-1124-13b-instruct", "model_route_id": "allenai__olmo-2-1124-13b-instruct", "model_name": "OLMo 2 13B Instruct November 2024", "developer": "allenai", "variant_key": "default", "raw_model_id": "allenai/olmo-2-1124-13b-instruct", "score": 0.81, "evaluation_id": "helm_safety/allenai_olmo-2-1124-13b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-13b-instruct/helm_safety_allenai_olmo_2_1124_13b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "allenai/olmo-2-1124-7b-instruct", "model_route_id": "allenai__olmo-2-1124-7b-instruct", "model_name": "OLMo 2 7B Instruct November 2024", "developer": "allenai", "variant_key": "default", "raw_model_id": "allenai/olmo-2-1124-7b-instruct", "score": 0.775, "evaluation_id": "helm_safety/allenai_olmo-2-1124-7b-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-7b-instruct/helm_safety_allenai_olmo_2_1124_7b_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "allenai/olmoe-1b-7b-0125-instruct", "model_route_id": "allenai__olmoe-1b-7b-0125-instruct", "model_name": "OLMoE 1B-7B Instruct January 2025", "developer": "allenai", "variant_key": "default", "raw_model_id": "allenai/olmoe-1b-7b-0125-instruct", "score": 0.725, "evaluation_id": "helm_safety/allenai_olmoe-1b-7b-0125-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmoe-1b-7b-0125-instruct/helm_safety_allenai_olmoe_1b_7b_0125_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "databricks/dbrx-instruct", "model_route_id": "databricks__dbrx-instruct", "model_name": "DBRX Instruct", "developer": "databricks", "variant_key": "default", "raw_model_id": "databricks/dbrx-instruct", "score": 0.535, "evaluation_id": "helm_safety/databricks_dbrx-instruct/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/databricks__dbrx-instruct/helm_safety_databricks_dbrx_instruct_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } }, { "model_id": "mistralai/mistral-7b-instruct-v0-1", "model_route_id": "mistralai__mistral-7b-instruct-v0-1", "model_name": "Mistral Instruct v0.1 7B", "developer": "mistralai", "variant_key": "default", "raw_model_id": "mistralai/mistral-7b-instruct-v0.1", "score": 0.432, "evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.1/1777076383.2276576", "retrieved_timestamp": "1777076383.2276576", "source_metadata": { "source_name": "helm_safety", "source_type": "documentation", "source_organization_name": "crfm", "evaluator_relationship": "third_party" }, "source_data": { "dataset_name": "helm_safety", "source_type": "url", "url": [ "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" ] }, "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-1/helm_safety_mistralai_mistral_7b_instruct_v0_1_1777076383_2276576.json", "detailed_evaluation_results": null, "detailed_evaluation_results_meta": null, "passthrough_top_level_fields": null, "instance_level_data": null, "normalized_result": { "benchmark_family_key": "helm_safety", "benchmark_family_name": "SimpleSafetyTests", "benchmark_parent_key": "helm_safety", "benchmark_parent_name": "SimpleSafetyTests", "benchmark_component_key": null, "benchmark_component_name": null, "benchmark_leaf_key": "simplesafetytests", "benchmark_leaf_name": "SimpleSafetyTests", "slice_key": null, "slice_name": null, "metric_name": "LM Evaluated Safety score", "metric_id": "lm_evaluated_safety_score", "metric_key": "lm_evaluated_safety_score", "metric_source": "evaluation_description", "display_name": "LM Evaluated Safety score", "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", "raw_evaluation_name": "SimpleSafetyTests", "is_summary_score": false }, "evalcards": { "annotations": { "reproducibility_gap": { "has_reproducibility_gap": true, "missing_fields": [ "temperature", "max_tokens" ], "required_field_count": 2, "populated_field_count": 0, "signal_version": "1.0" }, "provenance": { "source_type": "third_party", "is_multi_source": false, "first_party_only": false, "distinct_reporting_organizations": 1, "signal_version": "1.0" }, "variant_divergence": null, "cross_party_divergence": null } } } ], "models_count": 87, "top_score": 1.0 } ], "subtasks_count": 0, "metrics_count": 1, "models_count": 85, "metric_names": [ "LM Evaluated Safety score" ], "primary_metric_name": "LM Evaluated Safety score", "top_score": 1.0, "instance_data": { "available": false, "url_count": 0, "sample_urls": [], "models_with_loaded_instances": 0 }, "evalcards": { "annotations": { "reporting_completeness": { "completeness_score": 0.10714285714285714, "total_fields_evaluated": 28, "missing_required_fields": [ "autobenchmarkcard.benchmark_details.name", "autobenchmarkcard.benchmark_details.overview", "autobenchmarkcard.benchmark_details.data_type", "autobenchmarkcard.benchmark_details.domains", "autobenchmarkcard.benchmark_details.languages", "autobenchmarkcard.benchmark_details.similar_benchmarks", "autobenchmarkcard.benchmark_details.resources", "autobenchmarkcard.purpose_and_intended_users.goal", "autobenchmarkcard.purpose_and_intended_users.audience", "autobenchmarkcard.purpose_and_intended_users.tasks", "autobenchmarkcard.purpose_and_intended_users.limitations", "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "autobenchmarkcard.methodology.methods", "autobenchmarkcard.methodology.metrics", "autobenchmarkcard.methodology.calculation", "autobenchmarkcard.methodology.interpretation", "autobenchmarkcard.methodology.baseline_results", "autobenchmarkcard.methodology.validation", "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "autobenchmarkcard.data", "evalcards.lifecycle_status", "evalcards.preregistration_url" ], "partial_fields": [], "field_scores": [ { "field_path": "autobenchmarkcard.benchmark_details.name", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.overview", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.data_type", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.domains", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.languages", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.benchmark_details.resources", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.methods", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.metrics", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.calculation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.interpretation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.baseline_results", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.methodology.validation", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", "coverage_type": "full", "score": 0.0 }, { "field_path": "autobenchmarkcard.data", "coverage_type": "partial", "score": 0.0 }, { "field_path": "eee_eval.source_metadata.source_type", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.source_organization_name", "coverage_type": "full", "score": 1.0 }, { "field_path": "eee_eval.source_metadata.evaluator_relationship", "coverage_type": "full", "score": 1.0 }, { "field_path": "evalcards.lifecycle_status", "coverage_type": "reserved", "score": 0.0 }, { "field_path": "evalcards.preregistration_url", "coverage_type": "reserved", "score": 0.0 } ], "signal_version": "1.0" }, "benchmark_comparability": { "variant_divergence_groups": [], "cross_party_divergence_groups": [] } } }, "reproducibility_summary": { "results_total": 87, "has_reproducibility_gap_count": 87, "populated_ratio_avg": 0.0 }, "provenance_summary": { "total_results": 87, "total_groups": 85, "multi_source_groups": 0, "first_party_only_groups": 0, "source_type_distribution": { "first_party": 0, "third_party": 87, "collaborative": 0, "unspecified": 0 } }, "comparability_summary": { "total_groups": 85, "groups_with_variant_check": 0, "groups_with_cross_party_check": 0, "variant_divergent_count": 0, "cross_party_divergent_count": 0 } }