general-eval-card / tests /fixtures /evals /helm_safety_simplesafetytests.json
Jenny Chim
Add three-tier test infrastructure for migration safety
d3cbe09
Raw
History Blame
285 kB
{
"eval_summary_id": "helm_safety_simplesafetytests",
"benchmark": "SimpleSafetyTests",
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"evaluation_name": "SimpleSafetyTests",
"display_name": "SimpleSafetyTests",
"canonical_display_name": "SimpleSafetyTests",
"is_summary_score": false,
"category": "general",
"source_data": {
"dataset_name": "SimpleSafetyTests",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"benchmark_card": null,
"tags": {
"domains": [],
"languages": [],
"tasks": []
},
"subtasks": [],
"metrics": [
{
"metric_summary_id": "helm_safety_simplesafetytests_lm_evaluated_safety_score",
"legacy_eval_summary_id": "helm_safety_simplesafetytests",
"evaluation_name": "SimpleSafetyTests",
"display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"lower_is_better": false,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"metric_config": {
"evaluation_description": "LM Evaluated Safety score on SimpleSafetyTests",
"lower_is_better": false,
"score_type": "continuous",
"min_score": 0.0,
"max_score": 1.0
},
"model_results": [
{
"model_id": "writer/palmyra-x5",
"model_route_id": "writer__palmyra-x5",
"model_name": "Palmyra X5",
"developer": "writer",
"variant_key": "default",
"raw_model_id": "writer/palmyra-x5",
"score": 1.0,
"evaluation_id": "helm_safety/writer_palmyra-x5/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x5/helm_safety_writer_palmyra_x5_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "writer/palmyra-x-004",
"model_route_id": "writer__palmyra-x-004",
"model_name": "Palmyra-X-004",
"developer": "writer",
"variant_key": "default",
"raw_model_id": "writer/palmyra-x-004",
"score": 1.0,
"evaluation_id": "helm_safety/writer_palmyra-x-004/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-004/helm_safety_writer_palmyra_x_004_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "writer/palmyra-fin",
"model_route_id": "writer__palmyra-fin",
"model_name": "Palmyra Fin",
"developer": "writer",
"variant_key": "default",
"raw_model_id": "writer/palmyra-fin",
"score": 1.0,
"evaluation_id": "helm_safety/writer_palmyra-fin/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-fin/helm_safety_writer_palmyra_fin_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8",
"model_route_id": "qwen__qwen3-235b-a22b-instruct-2507-fp8",
"model_name": "Qwen3 235B A22B Instruct 2507 FP8",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8",
"score": 1.0,
"evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-instruct-2507-fp8/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-instruct-2507-fp8/helm_safety_qwen_qwen3_235b_a22b_instruct_2507_fp8_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen2-5-72b-instruct-turbo",
"model_route_id": "qwen__qwen2-5-72b-instruct-turbo",
"model_name": "Qwen2.5 Instruct Turbo 72B",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen2.5-72b-instruct-turbo",
"score": 1.0,
"evaluation_id": "helm_safety/qwen_qwen2.5-72b-instruct-turbo/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-72b-instruct-turbo/helm_safety_qwen_qwen2_5_72b_instruct_turbo_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/o4-mini",
"model_route_id": "openai__o4-mini",
"model_name": "o4-mini 2025-04-16",
"developer": "openai",
"variant_key": "2025-04-16",
"raw_model_id": "openai/o4-mini-2025-04-16",
"score": 1.0,
"evaluation_id": "helm_safety/openai_o4-mini-2025-04-16/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o4-mini/helm_safety_openai_o4_mini_2025_04_16_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-oss-20b",
"model_route_id": "openai__gpt-oss-20b",
"model_name": "gpt-oss-20b",
"developer": "openai",
"variant_key": "default",
"raw_model_id": "openai/gpt-oss-20b",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-oss-20b/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-20b/helm_safety_openai_gpt_oss_20b_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-oss-120b",
"model_route_id": "openai__gpt-oss-120b",
"model_name": "gpt-oss-120b",
"developer": "openai",
"variant_key": "default",
"raw_model_id": "openai/gpt-oss-120b",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-oss-120b/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-120b/helm_safety_openai_gpt_oss_120b_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-5-nano",
"model_route_id": "openai__gpt-5-nano",
"model_name": "GPT-5 nano 2025-08-07",
"developer": "openai",
"variant_key": "2025-08-07",
"raw_model_id": "openai/gpt-5-nano-2025-08-07",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-5-nano-2025-08-07/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-nano/helm_safety_openai_gpt_5_nano_2025_08_07_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-5-mini",
"model_route_id": "openai__gpt-5-mini",
"model_name": "GPT-5 mini 2025-08-07",
"developer": "openai",
"variant_key": "2025-08-07",
"raw_model_id": "openai/gpt-5-mini-2025-08-07",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-5-mini-2025-08-07/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-mini/helm_safety_openai_gpt_5_mini_2025_08_07_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4-5-preview",
"model_route_id": "openai__gpt-4-5-preview",
"model_name": "GPT-4.5 2025-02-27 preview",
"developer": "openai",
"variant_key": "2025-02-27",
"raw_model_id": "openai/gpt-4.5-preview-2025-02-27",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-4.5-preview-2025-02-27/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-5-preview/helm_safety_openai_gpt_4_5_preview_2025_02_27_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4-1-mini",
"model_route_id": "openai__gpt-4-1-mini",
"model_name": "GPT-4.1 mini 2025-04-14",
"developer": "openai",
"variant_key": "2025-04-14",
"raw_model_id": "openai/gpt-4.1-mini-2025-04-14",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-4.1-mini-2025-04-14/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-mini/helm_safety_openai_gpt_4_1_mini_2025_04_14_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4-1",
"model_route_id": "openai__gpt-4-1",
"model_name": "GPT-4.1 2025-04-14",
"developer": "openai",
"variant_key": "2025-04-14",
"raw_model_id": "openai/gpt-4.1-2025-04-14",
"score": 1.0,
"evaluation_id": "helm_safety/openai_gpt-4.1-2025-04-14/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1/helm_safety_openai_gpt_4_1_2025_04_14_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "moonshotai/kimi-k2-instruct",
"model_route_id": "moonshotai__kimi-k2-instruct",
"model_name": "Kimi K2 Instruct",
"developer": "moonshotai",
"variant_key": "default",
"raw_model_id": "moonshotai/kimi-k2-instruct",
"score": 1.0,
"evaluation_id": "helm_safety/moonshotai_kimi-k2-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/moonshotai__kimi-k2-instruct/helm_safety_moonshotai_kimi_k2_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "ibm/granite-4-0-micro-with-guardian",
"model_route_id": "ibm__granite-4-0-micro-with-guardian",
"model_name": "IBM Granite 4.0 Micro with guardian",
"developer": "ibm",
"variant_key": "default",
"raw_model_id": "ibm/granite-4.0-micro-with-guardian",
"score": 1.0,
"evaluation_id": "helm_safety/ibm_granite-4.0-micro-with-guardian/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro-with-guardian/helm_safety_ibm_granite_4_0_micro_with_guardian_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "ibm/granite-4-0-h-small-with-guardian",
"model_route_id": "ibm__granite-4-0-h-small-with-guardian",
"model_name": "IBM Granite 4.0 Small with guardian",
"developer": "ibm",
"variant_key": "default",
"raw_model_id": "ibm/granite-4.0-h-small-with-guardian",
"score": 1.0,
"evaluation_id": "helm_safety/ibm_granite-4.0-h-small-with-guardian/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small-with-guardian/helm_safety_ibm_granite_4_0_h_small_with_guardian_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "cohere/command-r-plus",
"model_route_id": "cohere__command-r-plus",
"model_name": "Command R Plus",
"developer": "cohere",
"variant_key": "default",
"raw_model_id": "cohere/command-r-plus",
"score": 1.0,
"evaluation_id": "helm_safety/cohere_command-r-plus/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r-plus/helm_safety_cohere_command_r_plus_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-sonnet-4-5",
"model_route_id": "anthropic__claude-sonnet-4-5",
"model_name": "Claude 4.5 Sonnet 20250929",
"developer": "anthropic",
"variant_key": "20250929",
"raw_model_id": "anthropic/claude-sonnet-4-5-20250929",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-sonnet-4-5-20250929/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4-5/helm_safety_anthropic_claude_sonnet_4_5_20250929_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-sonnet-4",
"model_route_id": "anthropic__claude-sonnet-4",
"model_name": "Claude 4 Sonnet 20250514, extended thinking",
"developer": "anthropic",
"variant_key": "20250514-thinking-10k",
"raw_model_id": "anthropic/claude-sonnet-4-20250514-thinking-10k",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514-thinking-10k/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_thinking_10k_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-opus-4",
"model_route_id": "anthropic__claude-opus-4",
"model_name": "Claude 4 Opus 20250514, extended thinking",
"developer": "anthropic",
"variant_key": "20250514-thinking-10k",
"raw_model_id": "anthropic/claude-opus-4-20250514-thinking-10k",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514-thinking-10k/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_thinking_10k_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-3-sonnet",
"model_route_id": "anthropic__claude-3-sonnet",
"model_name": "Claude 3 Sonnet 20240229",
"developer": "anthropic",
"variant_key": "20240229",
"raw_model_id": "anthropic/claude-3-sonnet-20240229",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-3-sonnet-20240229/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-sonnet/helm_safety_anthropic_claude_3_sonnet_20240229_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-3-opus",
"model_route_id": "anthropic__claude-3-opus",
"model_name": "Claude 3 Opus 20240229",
"developer": "anthropic",
"variant_key": "20240229",
"raw_model_id": "anthropic/claude-3-opus-20240229",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-3-opus-20240229/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-opus/helm_safety_anthropic_claude_3_opus_20240229_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-3-haiku",
"model_route_id": "anthropic__claude-3-haiku",
"model_name": "Claude 3 Haiku 20240307",
"developer": "anthropic",
"variant_key": "20240307",
"raw_model_id": "anthropic/claude-3-haiku-20240307",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-3-haiku-20240307/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-haiku/helm_safety_anthropic_claude_3_haiku_20240307_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-3-7-sonnet",
"model_route_id": "anthropic__claude-3-7-sonnet",
"model_name": "Claude 3.7 Sonnet 20250219",
"developer": "anthropic",
"variant_key": "20250219",
"raw_model_id": "anthropic/claude-3-7-sonnet-20250219",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-3-7-sonnet-20250219/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-7-sonnet/helm_safety_anthropic_claude_3_7_sonnet_20250219_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-3-5-sonnet",
"model_route_id": "anthropic__claude-3-5-sonnet",
"model_name": "Claude 3.5 Sonnet 20240620",
"developer": "anthropic",
"variant_key": "20240620",
"raw_model_id": "anthropic/claude-3-5-sonnet-20240620",
"score": 1.0,
"evaluation_id": "helm_safety/anthropic_claude-3-5-sonnet-20240620/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-sonnet/helm_safety_anthropic_claude_3_5_sonnet_20240620_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-5-1",
"model_route_id": "openai__gpt-5-1",
"model_name": "GPT-5.1 2025-11-13",
"developer": "openai",
"variant_key": "2025-11-13",
"raw_model_id": "openai/gpt-5.1-2025-11-13",
"score": 0.998,
"evaluation_id": "helm_safety/openai_gpt-5.1-2025-11-13/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-1/helm_safety_openai_gpt_5_1_2025_11_13_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-5",
"model_route_id": "openai__gpt-5",
"model_name": "GPT-5 2025-08-07",
"developer": "openai",
"variant_key": "2025-08-07",
"raw_model_id": "openai/gpt-5-2025-08-07",
"score": 0.998,
"evaluation_id": "helm_safety/openai_gpt-5-2025-08-07/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5/helm_safety_openai_gpt_5_2025_08_07_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen3-next-80b-a3b-thinking",
"model_route_id": "qwen__qwen3-next-80b-a3b-thinking",
"model_name": "Qwen3-Next 80B A3B Thinking",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen3-next-80b-a3b-thinking",
"score": 0.995,
"evaluation_id": "helm_safety/qwen_qwen3-next-80b-a3b-thinking/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-next-80b-a3b-thinking/helm_safety_qwen_qwen3_next_80b_a3b_thinking_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-sonnet-4",
"model_route_id": "anthropic__claude-sonnet-4",
"model_name": "Claude 4 Sonnet 20250514",
"developer": "anthropic",
"variant_key": "20250514",
"raw_model_id": "anthropic/claude-sonnet-4-20250514",
"score": 0.995,
"evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-opus-4",
"model_route_id": "anthropic__claude-opus-4",
"model_name": "Claude 4 Opus 20250514",
"developer": "anthropic",
"variant_key": "20250514",
"raw_model_id": "anthropic/claude-opus-4-20250514",
"score": 0.995,
"evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "xai/grok-3-mini-beta",
"model_route_id": "xai__grok-3-mini-beta",
"model_name": "Grok 3 mini Beta",
"developer": "xai",
"variant_key": "default",
"raw_model_id": "xai/grok-3-mini-beta",
"score": 0.993,
"evaluation_id": "helm_safety/xai_grok-3-mini-beta/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-mini-beta/helm_safety_xai_grok_3_mini_beta_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8",
"model_route_id": "meta__llama-4-maverick-17b-128e-instruct-fp8",
"model_name": "Llama 4 Maverick 17Bx128E Instruct FP8",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8",
"score": 0.993,
"evaluation_id": "helm_safety/meta_llama-4-maverick-17b-128e-instruct-fp8/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-maverick-17b-128e-instruct-fp8/helm_safety_meta_llama_4_maverick_17b_128e_instruct_fp8_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-3-8b-chat",
"model_route_id": "meta__llama-3-8b-chat",
"model_name": "Llama 3 Instruct 8B",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-3-8b-chat",
"score": 0.993,
"evaluation_id": "helm_safety/meta_llama-3-8b-chat/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-8b-chat/helm_safety_meta_llama_3_8b_chat_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "zai-org/glm-4-5-air-fp8",
"model_route_id": "zai-org__glm-4-5-air-fp8",
"model_name": "GLM-4.5-Air-FP8",
"developer": "zai-org",
"variant_key": "default",
"raw_model_id": "zai-org/glm-4.5-air-fp8",
"score": 0.99,
"evaluation_id": "helm_safety/zai-org_glm-4.5-air-fp8/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/zai-org__glm-4-5-air-fp8/helm_safety_zai_org_glm_4_5_air_fp8_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen1-5-72b-chat",
"model_route_id": "qwen__qwen1-5-72b-chat",
"model_name": "Qwen1.5 Chat 72B",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen1.5-72b-chat",
"score": 0.99,
"evaluation_id": "helm_safety/qwen_qwen1.5-72b-chat/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-72b-chat/helm_safety_qwen_qwen1_5_72b_chat_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/o3-mini",
"model_route_id": "openai__o3-mini",
"model_name": "o3-mini 2025-01-31",
"developer": "openai",
"variant_key": "2025-01-31",
"raw_model_id": "openai/o3-mini-2025-01-31",
"score": 0.99,
"evaluation_id": "helm_safety/openai_o3-mini-2025-01-31/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3-mini/helm_safety_openai_o3_mini_2025_01_31_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/o3",
"model_route_id": "openai__o3",
"model_name": "o3 2025-04-16",
"developer": "openai",
"variant_key": "2025-04-16",
"raw_model_id": "openai/o3-2025-04-16",
"score": 0.99,
"evaluation_id": "helm_safety/openai_o3-2025-04-16/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3/helm_safety_openai_o3_2025_04_16_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/o1",
"model_route_id": "openai__o1",
"model_name": "o1 2024-12-17",
"developer": "openai",
"variant_key": "2024-12-17",
"raw_model_id": "openai/o1-2024-12-17",
"score": 0.99,
"evaluation_id": "helm_safety/openai_o1-2024-12-17/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1/helm_safety_openai_o1_2024_12_17_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4-turbo",
"model_route_id": "openai__gpt-4-turbo",
"model_name": "GPT-4 Turbo 2024-04-09",
"developer": "openai",
"variant_key": "2024-04-09",
"raw_model_id": "openai/gpt-4-turbo-2024-04-09",
"score": 0.99,
"evaluation_id": "helm_safety/openai_gpt-4-turbo-2024-04-09/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-turbo/helm_safety_openai_gpt_4_turbo_2024_04_09_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4-1-nano",
"model_route_id": "openai__gpt-4-1-nano",
"model_name": "GPT-4.1 nano 2025-04-14",
"developer": "openai",
"variant_key": "2025-04-14",
"raw_model_id": "openai/gpt-4.1-nano-2025-04-14",
"score": 0.99,
"evaluation_id": "helm_safety/openai_gpt-4.1-nano-2025-04-14/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-nano/helm_safety_openai_gpt_4_1_nano_2025_04_14_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-3-70b-chat",
"model_route_id": "meta__llama-3-70b-chat",
"model_name": "Llama 3 Instruct 70B",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-3-70b-chat",
"score": 0.99,
"evaluation_id": "helm_safety/meta_llama-3-70b-chat/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-70b-chat/helm_safety_meta_llama_3_70b_chat_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-3-1-8b-instruct-turbo",
"model_route_id": "meta__llama-3-1-8b-instruct-turbo",
"model_name": "Llama 3.1 Instruct Turbo 8B",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-3.1-8b-instruct-turbo",
"score": 0.988,
"evaluation_id": "helm_safety/meta_llama-3.1-8b-instruct-turbo/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-8b-instruct-turbo/helm_safety_meta_llama_3_1_8b_instruct_turbo_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-3-1-405b-instruct-turbo",
"model_route_id": "meta__llama-3-1-405b-instruct-turbo",
"model_name": "Llama 3.1 Instruct Turbo 405B",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-3.1-405b-instruct-turbo",
"score": 0.988,
"evaluation_id": "helm_safety/meta_llama-3.1-405b-instruct-turbo/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-405b-instruct-turbo/helm_safety_meta_llama_3_1_405b_instruct_turbo_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "anthropic/claude-haiku-4-5",
"model_route_id": "anthropic__claude-haiku-4-5",
"model_name": "Claude 4.5 Haiku 20251001",
"developer": "anthropic",
"variant_key": "20251001",
"raw_model_id": "anthropic/claude-haiku-4-5-20251001",
"score": 0.988,
"evaluation_id": "helm_safety/anthropic_claude-haiku-4-5-20251001/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-haiku-4-5/helm_safety_anthropic_claude_haiku_4_5_20251001_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "writer/palmyra-med",
"model_route_id": "writer__palmyra-med",
"model_name": "Palmyra Med",
"developer": "writer",
"variant_key": "default",
"raw_model_id": "writer/palmyra-med",
"score": 0.985,
"evaluation_id": "helm_safety/writer_palmyra-med/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-med/helm_safety_writer_palmyra_med_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen3-235b-a22b-fp8-tput",
"model_route_id": "qwen__qwen3-235b-a22b-fp8-tput",
"model_name": "Qwen3 235B A22B FP8 Throughput",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen3-235b-a22b-fp8-tput",
"score": 0.985,
"evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-fp8-tput/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-fp8-tput/helm_safety_qwen_qwen3_235b_a22b_fp8_tput_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen2-72b-instruct",
"model_route_id": "qwen__qwen2-72b-instruct",
"model_name": "Qwen2 Instruct 72B",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen2-72b-instruct",
"score": 0.985,
"evaluation_id": "helm_safety/qwen_qwen2-72b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-72b-instruct/helm_safety_qwen_qwen2_72b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4o",
"model_route_id": "openai__gpt-4o",
"model_name": "GPT-4o 2024-05-13",
"developer": "openai",
"variant_key": "2024-05-13",
"raw_model_id": "openai/gpt-4o-2024-05-13",
"score": 0.985,
"evaluation_id": "helm_safety/openai_gpt-4o-2024-05-13/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o/helm_safety_openai_gpt_4o_2024_05_13_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-0-flash-001",
"model_route_id": "google__gemini-2-0-flash-001",
"model_name": "Gemini 2.0 Flash",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.0-flash-001",
"score": 0.985,
"evaluation_id": "helm_safety/google_gemini-2.0-flash-001/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-001/helm_safety_google_gemini_2_0_flash_001_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "deepseek-ai/deepseek-r1-0528",
"model_route_id": "deepseek-ai__deepseek-r1-0528",
"model_name": "DeepSeek-R1-0528",
"developer": "deepseek-ai",
"variant_key": "default",
"raw_model_id": "deepseek-ai/deepseek-r1-0528",
"score": 0.983,
"evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-0528/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-0528/helm_safety_deepseek_ai_deepseek_r1_0528_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "ibm/granite-4-0-h-small",
"model_route_id": "ibm__granite-4-0-h-small",
"model_name": "IBM Granite 4.0 Small",
"developer": "ibm",
"variant_key": "default",
"raw_model_id": "ibm/granite-4.0-h-small",
"score": 0.98,
"evaluation_id": "helm_safety/ibm_granite-4.0-h-small/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small/helm_safety_ibm_granite_4_0_h_small_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-5-flash-preview-04-17",
"model_route_id": "google__gemini-2-5-flash-preview-04-17",
"model_name": "Gemini 2.5 Flash 04-17 preview",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.5-flash-preview-04-17",
"score": 0.98,
"evaluation_id": "helm_safety/google_gemini-2.5-flash-preview-04-17/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-preview-04-17/helm_safety_google_gemini_2_5_flash_preview_04_17_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "deepseek-ai/deepseek-r1-hide-reasoning",
"model_route_id": "deepseek-ai__deepseek-r1-hide-reasoning",
"model_name": "DeepSeek R1 hide reasoning",
"developer": "deepseek-ai",
"variant_key": "default",
"raw_model_id": "deepseek-ai/deepseek-r1-hide-reasoning",
"score": 0.98,
"evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-hide-reasoning/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-hide-reasoning/helm_safety_deepseek_ai_deepseek_r1_hide_reasoning_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "allenai/olmo-2-0325-32b-instruct",
"model_route_id": "allenai__olmo-2-0325-32b-instruct",
"model_name": "OLMo 2 32B Instruct March 2025",
"developer": "allenai",
"variant_key": "default",
"raw_model_id": "allenai/olmo-2-0325-32b-instruct",
"score": 0.98,
"evaluation_id": "helm_safety/allenai_olmo-2-0325-32b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-0325-32b-instruct/helm_safety_allenai_olmo_2_0325_32b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-4o-mini",
"model_route_id": "openai__gpt-4o-mini",
"model_name": "GPT-4o mini 2024-07-18",
"developer": "openai",
"variant_key": "2024-07-18",
"raw_model_id": "openai/gpt-4o-mini-2024-07-18",
"score": 0.978,
"evaluation_id": "helm_safety/openai_gpt-4o-mini-2024-07-18/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o-mini/helm_safety_openai_gpt_4o_mini_2024_07_18_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-0-flash-lite-preview-02-05",
"model_route_id": "google__gemini-2-0-flash-lite-preview-02-05",
"model_name": "Gemini 2.0 Flash Lite 02-05 preview",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.0-flash-lite-preview-02-05",
"score": 0.977,
"evaluation_id": "helm_safety/google_gemini-2.0-flash-lite-preview-02-05/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-lite-preview-02-05/helm_safety_google_gemini_2_0_flash_lite_preview_02_05_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "ibm/granite-3-3-8b-instruct",
"model_route_id": "ibm__granite-3-3-8b-instruct",
"model_name": "IBM Granite 3.3 8B Instruct",
"developer": "ibm",
"variant_key": "default",
"raw_model_id": "ibm/granite-3.3-8b-instruct",
"score": 0.975,
"evaluation_id": "helm_safety/ibm_granite-3.3-8b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-3-3-8b-instruct/helm_safety_ibm_granite_3_3_8b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-3-pro-preview",
"model_route_id": "google__gemini-3-pro-preview",
"model_name": "Gemini 3 Pro Preview",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-3-pro-preview",
"score": 0.975,
"evaluation_id": "helm_safety/google_gemini-3-pro-preview/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-pro-preview/helm_safety_google_gemini_3_pro_preview_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-0-pro-exp-02-05",
"model_route_id": "google__gemini-2-0-pro-exp-02-05",
"model_name": "Gemini 2.0 Pro 02-05 preview",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.0-pro-exp-02-05",
"score": 0.975,
"evaluation_id": "helm_safety/google_gemini-2.0-pro-exp-02-05/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-pro-exp-02-05/helm_safety_google_gemini_2_0_pro_exp_02_05_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-1-5-pro-001",
"model_route_id": "google__gemini-1-5-pro-001",
"model_name": "Gemini 1.5 Pro 001",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-1.5-pro-001",
"score": 0.975,
"evaluation_id": "helm_safety/google_gemini-1.5-pro-001/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-pro-001/helm_safety_google_gemini_1_5_pro_001_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "deepseek-ai/deepseek-r1",
"model_route_id": "deepseek-ai__deepseek-r1",
"model_name": "DeepSeek R1",
"developer": "deepseek-ai",
"variant_key": "default",
"raw_model_id": "deepseek-ai/deepseek-r1",
"score": 0.975,
"evaluation_id": "helm_safety/deepseek-ai_deepseek-r1/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1/helm_safety_deepseek_ai_deepseek_r1_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/o1-mini",
"model_route_id": "openai__o1-mini",
"model_name": "o1-mini 2024-09-12",
"developer": "openai",
"variant_key": "2024-09-12",
"raw_model_id": "openai/o1-mini-2024-09-12",
"score": 0.97,
"evaluation_id": "helm_safety/openai_o1-mini-2024-09-12/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1-mini/helm_safety_openai_o1_mini_2024_09_12_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-4-scout-17b-16e-instruct",
"model_route_id": "meta__llama-4-scout-17b-16e-instruct",
"model_name": "Llama 4 Scout 17Bx16E Instruct",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-4-scout-17b-16e-instruct",
"score": 0.97,
"evaluation_id": "helm_safety/meta_llama-4-scout-17b-16e-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-scout-17b-16e-instruct/helm_safety_meta_llama_4_scout_17b_16e_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-5-pro-preview-03-25",
"model_route_id": "google__gemini-2-5-pro-preview-03-25",
"model_name": "Gemini 2.5 Pro 03-25 preview",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.5-pro-preview-03-25",
"score": 0.97,
"evaluation_id": "helm_safety/google_gemini-2.5-pro-preview-03-25/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-pro-preview-03-25/helm_safety_google_gemini_2_5_pro_preview_03_25_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-1-5-flash-001",
"model_route_id": "google__gemini-1-5-flash-001",
"model_name": "Gemini 1.5 Flash 001",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-1.5-flash-001",
"score": 0.97,
"evaluation_id": "helm_safety/google_gemini-1.5-flash-001/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-flash-001/helm_safety_google_gemini_1_5_flash_001_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "xai/grok-3-beta",
"model_route_id": "xai__grok-3-beta",
"model_name": "Grok 3 Beta",
"developer": "xai",
"variant_key": "default",
"raw_model_id": "xai/grok-3-beta",
"score": 0.968,
"evaluation_id": "helm_safety/xai_grok-3-beta/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-beta/helm_safety_xai_grok_3_beta_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "deepseek-ai/deepseek-llm-67b-chat",
"model_route_id": "deepseek-ai__deepseek-llm-67b-chat",
"model_name": "DeepSeek LLM Chat 67B",
"developer": "deepseek-ai",
"variant_key": "default",
"raw_model_id": "deepseek-ai/deepseek-llm-67b-chat",
"score": 0.968,
"evaluation_id": "helm_safety/deepseek-ai_deepseek-llm-67b-chat/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-llm-67b-chat/helm_safety_deepseek_ai_deepseek_llm_67b_chat_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "google/gemini-2-5-flash-lite",
"model_route_id": "google__gemini-2-5-flash-lite",
"model_name": "Gemini 2.5 Flash-Lite",
"developer": "google",
"variant_key": "default",
"raw_model_id": "google/gemini-2.5-flash-lite",
"score": 0.965,
"evaluation_id": "helm_safety/google_gemini-2.5-flash-lite/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-lite/helm_safety_google_gemini_2_5_flash_lite_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "qwen/qwen2-5-7b-instruct-turbo",
"model_route_id": "qwen__qwen2-5-7b-instruct-turbo",
"model_name": "Qwen2.5 Instruct Turbo 7B",
"developer": "qwen",
"variant_key": "default",
"raw_model_id": "qwen/qwen2.5-7b-instruct-turbo",
"score": 0.96,
"evaluation_id": "helm_safety/qwen_qwen2.5-7b-instruct-turbo/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-7b-instruct-turbo/helm_safety_qwen_qwen2_5_7b_instruct_turbo_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-3-5-turbo-0613",
"model_route_id": "openai__gpt-3-5-turbo-0613",
"model_name": "GPT-3.5 Turbo 0613",
"developer": "openai",
"variant_key": "default",
"raw_model_id": "openai/gpt-3.5-turbo-0613",
"score": 0.958,
"evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0613/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0613/helm_safety_openai_gpt_3_5_turbo_0613_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "marin-community/marin-8b-instruct",
"model_route_id": "marin-community__marin-8b-instruct",
"model_name": "Marin 8B Instruct",
"developer": "marin-community",
"variant_key": "default",
"raw_model_id": "marin-community/marin-8b-instruct",
"score": 0.958,
"evaluation_id": "helm_safety/marin-community_marin-8b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/marin-community__marin-8b-instruct/helm_safety_marin_community_marin_8b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "deepseek-ai/deepseek-v3",
"model_route_id": "deepseek-ai__deepseek-v3",
"model_name": "DeepSeek v3",
"developer": "deepseek-ai",
"variant_key": "default",
"raw_model_id": "deepseek-ai/deepseek-v3",
"score": 0.953,
"evaluation_id": "helm_safety/deepseek-ai_deepseek-v3/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-v3/helm_safety_deepseek_ai_deepseek_v3_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "ibm/granite-4-0-micro",
"model_route_id": "ibm__granite-4-0-micro",
"model_name": "IBM Granite 4.0 Micro",
"developer": "ibm",
"variant_key": "default",
"raw_model_id": "ibm/granite-4.0-micro",
"score": 0.945,
"evaluation_id": "helm_safety/ibm_granite-4.0-micro/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro/helm_safety_ibm_granite_4_0_micro_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "cohere/command-r",
"model_route_id": "cohere__command-r",
"model_name": "Command R",
"developer": "cohere",
"variant_key": "default",
"raw_model_id": "cohere/command-r",
"score": 0.943,
"evaluation_id": "helm_safety/cohere_command-r/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r/helm_safety_cohere_command_r_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "mistralai/mistral-small-2501",
"model_route_id": "mistralai__mistral-small-2501",
"model_name": "Mistral Small 3 2501",
"developer": "mistralai",
"variant_key": "default",
"raw_model_id": "mistralai/mistral-small-2501",
"score": 0.932,
"evaluation_id": "helm_safety/mistralai_mistral-small-2501/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-small-2501/helm_safety_mistralai_mistral_small_2501_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "meta/llama-3-1-70b-instruct-turbo",
"model_route_id": "meta__llama-3-1-70b-instruct-turbo",
"model_name": "Llama 3.1 Instruct Turbo 70B",
"developer": "meta",
"variant_key": "default",
"raw_model_id": "meta/llama-3.1-70b-instruct-turbo",
"score": 0.925,
"evaluation_id": "helm_safety/meta_llama-3.1-70b-instruct-turbo/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-70b-instruct-turbo/helm_safety_meta_llama_3_1_70b_instruct_turbo_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "xai/grok-4-0709",
"model_route_id": "xai__grok-4-0709",
"model_name": "Grok 4 0709",
"developer": "xai",
"variant_key": "default",
"raw_model_id": "xai/grok-4-0709",
"score": 0.922,
"evaluation_id": "helm_safety/xai_grok-4-0709/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-4-0709/helm_safety_xai_grok_4_0709_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-3-5-turbo-1106",
"model_route_id": "openai__gpt-3-5-turbo-1106",
"model_name": "GPT-3.5 Turbo 1106",
"developer": "openai",
"variant_key": "default",
"raw_model_id": "openai/gpt-3.5-turbo-1106",
"score": 0.922,
"evaluation_id": "helm_safety/openai_gpt-3.5-turbo-1106/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-1106/helm_safety_openai_gpt_3_5_turbo_1106_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "openai/gpt-3-5-turbo-0125",
"model_route_id": "openai__gpt-3-5-turbo-0125",
"model_name": "GPT-3.5 Turbo 0125",
"developer": "openai",
"variant_key": "default",
"raw_model_id": "openai/gpt-3.5-turbo-0125",
"score": 0.92,
"evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0125/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0125/helm_safety_openai_gpt_3_5_turbo_0125_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "mistralai/mixtral-8x22b-instruct-v0-1",
"model_route_id": "mistralai__mixtral-8x22b-instruct-v0-1",
"model_name": "Mixtral Instruct 8x22B",
"developer": "mistralai",
"variant_key": "default",
"raw_model_id": "mistralai/mixtral-8x22b-instruct-v0.1",
"score": 0.915,
"evaluation_id": "helm_safety/mistralai_mixtral-8x22b-instruct-v0.1/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x22b-instruct-v0-1/helm_safety_mistralai_mixtral_8x22b_instruct_v0_1_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "mistralai/mixtral-8x7b-instruct-v0-1",
"model_route_id": "mistralai__mixtral-8x7b-instruct-v0-1",
"model_name": "Mixtral Instruct 8x7B",
"developer": "mistralai",
"variant_key": "default",
"raw_model_id": "mistralai/mixtral-8x7b-instruct-v0.1",
"score": 0.905,
"evaluation_id": "helm_safety/mistralai_mixtral-8x7b-instruct-v0.1/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x7b-instruct-v0-1/helm_safety_mistralai_mixtral_8x7b_instruct_v0_1_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "mistralai/mistral-7b-instruct-v0-3",
"model_route_id": "mistralai__mistral-7b-instruct-v0-3",
"model_name": "Mistral Instruct v0.3 7B",
"developer": "mistralai",
"variant_key": "default",
"raw_model_id": "mistralai/mistral-7b-instruct-v0.3",
"score": 0.81,
"evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.3/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-3/helm_safety_mistralai_mistral_7b_instruct_v0_3_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "allenai/olmo-2-1124-13b-instruct",
"model_route_id": "allenai__olmo-2-1124-13b-instruct",
"model_name": "OLMo 2 13B Instruct November 2024",
"developer": "allenai",
"variant_key": "default",
"raw_model_id": "allenai/olmo-2-1124-13b-instruct",
"score": 0.81,
"evaluation_id": "helm_safety/allenai_olmo-2-1124-13b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-13b-instruct/helm_safety_allenai_olmo_2_1124_13b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "allenai/olmo-2-1124-7b-instruct",
"model_route_id": "allenai__olmo-2-1124-7b-instruct",
"model_name": "OLMo 2 7B Instruct November 2024",
"developer": "allenai",
"variant_key": "default",
"raw_model_id": "allenai/olmo-2-1124-7b-instruct",
"score": 0.775,
"evaluation_id": "helm_safety/allenai_olmo-2-1124-7b-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-7b-instruct/helm_safety_allenai_olmo_2_1124_7b_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "allenai/olmoe-1b-7b-0125-instruct",
"model_route_id": "allenai__olmoe-1b-7b-0125-instruct",
"model_name": "OLMoE 1B-7B Instruct January 2025",
"developer": "allenai",
"variant_key": "default",
"raw_model_id": "allenai/olmoe-1b-7b-0125-instruct",
"score": 0.725,
"evaluation_id": "helm_safety/allenai_olmoe-1b-7b-0125-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmoe-1b-7b-0125-instruct/helm_safety_allenai_olmoe_1b_7b_0125_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "databricks/dbrx-instruct",
"model_route_id": "databricks__dbrx-instruct",
"model_name": "DBRX Instruct",
"developer": "databricks",
"variant_key": "default",
"raw_model_id": "databricks/dbrx-instruct",
"score": 0.535,
"evaluation_id": "helm_safety/databricks_dbrx-instruct/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/databricks__dbrx-instruct/helm_safety_databricks_dbrx_instruct_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
},
{
"model_id": "mistralai/mistral-7b-instruct-v0-1",
"model_route_id": "mistralai__mistral-7b-instruct-v0-1",
"model_name": "Mistral Instruct v0.1 7B",
"developer": "mistralai",
"variant_key": "default",
"raw_model_id": "mistralai/mistral-7b-instruct-v0.1",
"score": 0.432,
"evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.1/1777076383.2276576",
"retrieved_timestamp": "1777076383.2276576",
"source_metadata": {
"source_name": "helm_safety",
"source_type": "documentation",
"source_organization_name": "crfm",
"evaluator_relationship": "third_party"
},
"source_data": {
"dataset_name": "helm_safety",
"source_type": "url",
"url": [
"https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json"
]
},
"source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-1/helm_safety_mistralai_mistral_7b_instruct_v0_1_1777076383_2276576.json",
"detailed_evaluation_results": null,
"detailed_evaluation_results_meta": null,
"passthrough_top_level_fields": null,
"instance_level_data": null,
"normalized_result": {
"benchmark_family_key": "helm_safety",
"benchmark_family_name": "SimpleSafetyTests",
"benchmark_parent_key": "helm_safety",
"benchmark_parent_name": "SimpleSafetyTests",
"benchmark_component_key": null,
"benchmark_component_name": null,
"benchmark_leaf_key": "simplesafetytests",
"benchmark_leaf_name": "SimpleSafetyTests",
"slice_key": null,
"slice_name": null,
"metric_name": "LM Evaluated Safety score",
"metric_id": "lm_evaluated_safety_score",
"metric_key": "lm_evaluated_safety_score",
"metric_source": "evaluation_description",
"display_name": "LM Evaluated Safety score",
"canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score",
"raw_evaluation_name": "SimpleSafetyTests",
"is_summary_score": false
},
"evalcards": {
"annotations": {
"reproducibility_gap": {
"has_reproducibility_gap": true,
"missing_fields": [
"temperature",
"max_tokens"
],
"required_field_count": 2,
"populated_field_count": 0,
"signal_version": "1.0"
},
"provenance": {
"source_type": "third_party",
"is_multi_source": false,
"first_party_only": false,
"distinct_reporting_organizations": 1,
"signal_version": "1.0"
},
"variant_divergence": null,
"cross_party_divergence": null
}
}
}
],
"models_count": 87,
"top_score": 1.0
}
],
"subtasks_count": 0,
"metrics_count": 1,
"models_count": 85,
"metric_names": [
"LM Evaluated Safety score"
],
"primary_metric_name": "LM Evaluated Safety score",
"top_score": 1.0,
"instance_data": {
"available": false,
"url_count": 0,
"sample_urls": [],
"models_with_loaded_instances": 0
},
"evalcards": {
"annotations": {
"reporting_completeness": {
"completeness_score": 0.10714285714285714,
"total_fields_evaluated": 28,
"missing_required_fields": [
"autobenchmarkcard.benchmark_details.name",
"autobenchmarkcard.benchmark_details.overview",
"autobenchmarkcard.benchmark_details.data_type",
"autobenchmarkcard.benchmark_details.domains",
"autobenchmarkcard.benchmark_details.languages",
"autobenchmarkcard.benchmark_details.similar_benchmarks",
"autobenchmarkcard.benchmark_details.resources",
"autobenchmarkcard.purpose_and_intended_users.goal",
"autobenchmarkcard.purpose_and_intended_users.audience",
"autobenchmarkcard.purpose_and_intended_users.tasks",
"autobenchmarkcard.purpose_and_intended_users.limitations",
"autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses",
"autobenchmarkcard.methodology.methods",
"autobenchmarkcard.methodology.metrics",
"autobenchmarkcard.methodology.calculation",
"autobenchmarkcard.methodology.interpretation",
"autobenchmarkcard.methodology.baseline_results",
"autobenchmarkcard.methodology.validation",
"autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity",
"autobenchmarkcard.ethical_and_legal_considerations.data_licensing",
"autobenchmarkcard.ethical_and_legal_considerations.consent_procedures",
"autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations",
"autobenchmarkcard.data",
"evalcards.lifecycle_status",
"evalcards.preregistration_url"
],
"partial_fields": [],
"field_scores": [
{
"field_path": "autobenchmarkcard.benchmark_details.name",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.overview",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.data_type",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.domains",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.languages",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.benchmark_details.resources",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.purpose_and_intended_users.goal",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.purpose_and_intended_users.audience",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.purpose_and_intended_users.tasks",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.purpose_and_intended_users.limitations",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.methods",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.metrics",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.calculation",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.interpretation",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.baseline_results",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.methodology.validation",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations",
"coverage_type": "full",
"score": 0.0
},
{
"field_path": "autobenchmarkcard.data",
"coverage_type": "partial",
"score": 0.0
},
{
"field_path": "eee_eval.source_metadata.source_type",
"coverage_type": "full",
"score": 1.0
},
{
"field_path": "eee_eval.source_metadata.source_organization_name",
"coverage_type": "full",
"score": 1.0
},
{
"field_path": "eee_eval.source_metadata.evaluator_relationship",
"coverage_type": "full",
"score": 1.0
},
{
"field_path": "evalcards.lifecycle_status",
"coverage_type": "reserved",
"score": 0.0
},
{
"field_path": "evalcards.preregistration_url",
"coverage_type": "reserved",
"score": 0.0
}
],
"signal_version": "1.0"
},
"benchmark_comparability": {
"variant_divergence_groups": [],
"cross_party_divergence_groups": []
}
}
},
"reproducibility_summary": {
"results_total": 87,
"has_reproducibility_gap_count": 87,
"populated_ratio_avg": 0.0
},
"provenance_summary": {
"total_results": 87,
"total_groups": 85,
"multi_source_groups": 0,
"first_party_only_groups": 0,
"source_type_distribution": {
"first_party": 0,
"third_party": 87,
"collaborative": 0,
"unspecified": 0
}
},
"comparability_summary": {
"total_groups": 85,
"groups_with_variant_check": 0,
"groups_with_cross_party_check": 0,
"variant_divergent_count": 0,
"cross_party_divergent_count": 0
}
}