Spaces:
Running
Running
| { | |
| "eval_summary_id": "helm_safety_simplesafetytests", | |
| "benchmark": "SimpleSafetyTests", | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "evaluation_name": "SimpleSafetyTests", | |
| "display_name": "SimpleSafetyTests", | |
| "canonical_display_name": "SimpleSafetyTests", | |
| "is_summary_score": false, | |
| "category": "general", | |
| "source_data": { | |
| "dataset_name": "SimpleSafetyTests", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "benchmark_card": null, | |
| "tags": { | |
| "domains": [], | |
| "languages": [], | |
| "tasks": [] | |
| }, | |
| "subtasks": [], | |
| "metrics": [ | |
| { | |
| "metric_summary_id": "helm_safety_simplesafetytests_lm_evaluated_safety_score", | |
| "legacy_eval_summary_id": "helm_safety_simplesafetytests", | |
| "evaluation_name": "SimpleSafetyTests", | |
| "display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "lower_is_better": false, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "metric_config": { | |
| "evaluation_description": "LM Evaluated Safety score on SimpleSafetyTests", | |
| "lower_is_better": false, | |
| "score_type": "continuous", | |
| "min_score": 0.0, | |
| "max_score": 1.0 | |
| }, | |
| "model_results": [ | |
| { | |
| "model_id": "writer/palmyra-x5", | |
| "model_route_id": "writer__palmyra-x5", | |
| "model_name": "Palmyra X5", | |
| "developer": "writer", | |
| "variant_key": "default", | |
| "raw_model_id": "writer/palmyra-x5", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/writer_palmyra-x5/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x5/helm_safety_writer_palmyra_x5_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "writer/palmyra-x-004", | |
| "model_route_id": "writer__palmyra-x-004", | |
| "model_name": "Palmyra-X-004", | |
| "developer": "writer", | |
| "variant_key": "default", | |
| "raw_model_id": "writer/palmyra-x-004", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/writer_palmyra-x-004/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-x-004/helm_safety_writer_palmyra_x_004_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "writer/palmyra-fin", | |
| "model_route_id": "writer__palmyra-fin", | |
| "model_name": "Palmyra Fin", | |
| "developer": "writer", | |
| "variant_key": "default", | |
| "raw_model_id": "writer/palmyra-fin", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/writer_palmyra-fin/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-fin/helm_safety_writer_palmyra_fin_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8", | |
| "model_route_id": "qwen__qwen3-235b-a22b-instruct-2507-fp8", | |
| "model_name": "Qwen3 235B A22B Instruct 2507 FP8", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen3-235b-a22b-instruct-2507-fp8", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-instruct-2507-fp8/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-instruct-2507-fp8/helm_safety_qwen_qwen3_235b_a22b_instruct_2507_fp8_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen2-5-72b-instruct-turbo", | |
| "model_route_id": "qwen__qwen2-5-72b-instruct-turbo", | |
| "model_name": "Qwen2.5 Instruct Turbo 72B", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen2.5-72b-instruct-turbo", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/qwen_qwen2.5-72b-instruct-turbo/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-72b-instruct-turbo/helm_safety_qwen_qwen2_5_72b_instruct_turbo_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/o4-mini", | |
| "model_route_id": "openai__o4-mini", | |
| "model_name": "o4-mini 2025-04-16", | |
| "developer": "openai", | |
| "variant_key": "2025-04-16", | |
| "raw_model_id": "openai/o4-mini-2025-04-16", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_o4-mini-2025-04-16/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o4-mini/helm_safety_openai_o4_mini_2025_04_16_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-oss-20b", | |
| "model_route_id": "openai__gpt-oss-20b", | |
| "model_name": "gpt-oss-20b", | |
| "developer": "openai", | |
| "variant_key": "default", | |
| "raw_model_id": "openai/gpt-oss-20b", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-oss-20b/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-20b/helm_safety_openai_gpt_oss_20b_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-oss-120b", | |
| "model_route_id": "openai__gpt-oss-120b", | |
| "model_name": "gpt-oss-120b", | |
| "developer": "openai", | |
| "variant_key": "default", | |
| "raw_model_id": "openai/gpt-oss-120b", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-oss-120b/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-oss-120b/helm_safety_openai_gpt_oss_120b_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-5-nano", | |
| "model_route_id": "openai__gpt-5-nano", | |
| "model_name": "GPT-5 nano 2025-08-07", | |
| "developer": "openai", | |
| "variant_key": "2025-08-07", | |
| "raw_model_id": "openai/gpt-5-nano-2025-08-07", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-5-nano-2025-08-07/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-nano/helm_safety_openai_gpt_5_nano_2025_08_07_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-5-mini", | |
| "model_route_id": "openai__gpt-5-mini", | |
| "model_name": "GPT-5 mini 2025-08-07", | |
| "developer": "openai", | |
| "variant_key": "2025-08-07", | |
| "raw_model_id": "openai/gpt-5-mini-2025-08-07", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-5-mini-2025-08-07/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-mini/helm_safety_openai_gpt_5_mini_2025_08_07_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4-5-preview", | |
| "model_route_id": "openai__gpt-4-5-preview", | |
| "model_name": "GPT-4.5 2025-02-27 preview", | |
| "developer": "openai", | |
| "variant_key": "2025-02-27", | |
| "raw_model_id": "openai/gpt-4.5-preview-2025-02-27", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-4.5-preview-2025-02-27/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-5-preview/helm_safety_openai_gpt_4_5_preview_2025_02_27_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4-1-mini", | |
| "model_route_id": "openai__gpt-4-1-mini", | |
| "model_name": "GPT-4.1 mini 2025-04-14", | |
| "developer": "openai", | |
| "variant_key": "2025-04-14", | |
| "raw_model_id": "openai/gpt-4.1-mini-2025-04-14", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-4.1-mini-2025-04-14/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-mini/helm_safety_openai_gpt_4_1_mini_2025_04_14_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4-1", | |
| "model_route_id": "openai__gpt-4-1", | |
| "model_name": "GPT-4.1 2025-04-14", | |
| "developer": "openai", | |
| "variant_key": "2025-04-14", | |
| "raw_model_id": "openai/gpt-4.1-2025-04-14", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/openai_gpt-4.1-2025-04-14/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1/helm_safety_openai_gpt_4_1_2025_04_14_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "moonshotai/kimi-k2-instruct", | |
| "model_route_id": "moonshotai__kimi-k2-instruct", | |
| "model_name": "Kimi K2 Instruct", | |
| "developer": "moonshotai", | |
| "variant_key": "default", | |
| "raw_model_id": "moonshotai/kimi-k2-instruct", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/moonshotai_kimi-k2-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/moonshotai__kimi-k2-instruct/helm_safety_moonshotai_kimi_k2_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "ibm/granite-4-0-micro-with-guardian", | |
| "model_route_id": "ibm__granite-4-0-micro-with-guardian", | |
| "model_name": "IBM Granite 4.0 Micro with guardian", | |
| "developer": "ibm", | |
| "variant_key": "default", | |
| "raw_model_id": "ibm/granite-4.0-micro-with-guardian", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/ibm_granite-4.0-micro-with-guardian/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro-with-guardian/helm_safety_ibm_granite_4_0_micro_with_guardian_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "ibm/granite-4-0-h-small-with-guardian", | |
| "model_route_id": "ibm__granite-4-0-h-small-with-guardian", | |
| "model_name": "IBM Granite 4.0 Small with guardian", | |
| "developer": "ibm", | |
| "variant_key": "default", | |
| "raw_model_id": "ibm/granite-4.0-h-small-with-guardian", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/ibm_granite-4.0-h-small-with-guardian/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small-with-guardian/helm_safety_ibm_granite_4_0_h_small_with_guardian_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "cohere/command-r-plus", | |
| "model_route_id": "cohere__command-r-plus", | |
| "model_name": "Command R Plus", | |
| "developer": "cohere", | |
| "variant_key": "default", | |
| "raw_model_id": "cohere/command-r-plus", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/cohere_command-r-plus/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r-plus/helm_safety_cohere_command_r_plus_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-sonnet-4-5", | |
| "model_route_id": "anthropic__claude-sonnet-4-5", | |
| "model_name": "Claude 4.5 Sonnet 20250929", | |
| "developer": "anthropic", | |
| "variant_key": "20250929", | |
| "raw_model_id": "anthropic/claude-sonnet-4-5-20250929", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-5-20250929/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4-5/helm_safety_anthropic_claude_sonnet_4_5_20250929_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-sonnet-4", | |
| "model_route_id": "anthropic__claude-sonnet-4", | |
| "model_name": "Claude 4 Sonnet 20250514, extended thinking", | |
| "developer": "anthropic", | |
| "variant_key": "20250514-thinking-10k", | |
| "raw_model_id": "anthropic/claude-sonnet-4-20250514-thinking-10k", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514-thinking-10k/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_thinking_10k_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-opus-4", | |
| "model_route_id": "anthropic__claude-opus-4", | |
| "model_name": "Claude 4 Opus 20250514, extended thinking", | |
| "developer": "anthropic", | |
| "variant_key": "20250514-thinking-10k", | |
| "raw_model_id": "anthropic/claude-opus-4-20250514-thinking-10k", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514-thinking-10k/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_thinking_10k_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-3-sonnet", | |
| "model_route_id": "anthropic__claude-3-sonnet", | |
| "model_name": "Claude 3 Sonnet 20240229", | |
| "developer": "anthropic", | |
| "variant_key": "20240229", | |
| "raw_model_id": "anthropic/claude-3-sonnet-20240229", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-3-sonnet-20240229/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-sonnet/helm_safety_anthropic_claude_3_sonnet_20240229_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-3-opus", | |
| "model_route_id": "anthropic__claude-3-opus", | |
| "model_name": "Claude 3 Opus 20240229", | |
| "developer": "anthropic", | |
| "variant_key": "20240229", | |
| "raw_model_id": "anthropic/claude-3-opus-20240229", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-3-opus-20240229/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-opus/helm_safety_anthropic_claude_3_opus_20240229_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-3-haiku", | |
| "model_route_id": "anthropic__claude-3-haiku", | |
| "model_name": "Claude 3 Haiku 20240307", | |
| "developer": "anthropic", | |
| "variant_key": "20240307", | |
| "raw_model_id": "anthropic/claude-3-haiku-20240307", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-3-haiku-20240307/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-haiku/helm_safety_anthropic_claude_3_haiku_20240307_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-3-7-sonnet", | |
| "model_route_id": "anthropic__claude-3-7-sonnet", | |
| "model_name": "Claude 3.7 Sonnet 20250219", | |
| "developer": "anthropic", | |
| "variant_key": "20250219", | |
| "raw_model_id": "anthropic/claude-3-7-sonnet-20250219", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-3-7-sonnet-20250219/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-7-sonnet/helm_safety_anthropic_claude_3_7_sonnet_20250219_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-3-5-sonnet", | |
| "model_route_id": "anthropic__claude-3-5-sonnet", | |
| "model_name": "Claude 3.5 Sonnet 20240620", | |
| "developer": "anthropic", | |
| "variant_key": "20240620", | |
| "raw_model_id": "anthropic/claude-3-5-sonnet-20240620", | |
| "score": 1.0, | |
| "evaluation_id": "helm_safety/anthropic_claude-3-5-sonnet-20240620/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-3-5-sonnet/helm_safety_anthropic_claude_3_5_sonnet_20240620_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-5-1", | |
| "model_route_id": "openai__gpt-5-1", | |
| "model_name": "GPT-5.1 2025-11-13", | |
| "developer": "openai", | |
| "variant_key": "2025-11-13", | |
| "raw_model_id": "openai/gpt-5.1-2025-11-13", | |
| "score": 0.998, | |
| "evaluation_id": "helm_safety/openai_gpt-5.1-2025-11-13/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5-1/helm_safety_openai_gpt_5_1_2025_11_13_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-5", | |
| "model_route_id": "openai__gpt-5", | |
| "model_name": "GPT-5 2025-08-07", | |
| "developer": "openai", | |
| "variant_key": "2025-08-07", | |
| "raw_model_id": "openai/gpt-5-2025-08-07", | |
| "score": 0.998, | |
| "evaluation_id": "helm_safety/openai_gpt-5-2025-08-07/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-5/helm_safety_openai_gpt_5_2025_08_07_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen3-next-80b-a3b-thinking", | |
| "model_route_id": "qwen__qwen3-next-80b-a3b-thinking", | |
| "model_name": "Qwen3-Next 80B A3B Thinking", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen3-next-80b-a3b-thinking", | |
| "score": 0.995, | |
| "evaluation_id": "helm_safety/qwen_qwen3-next-80b-a3b-thinking/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-next-80b-a3b-thinking/helm_safety_qwen_qwen3_next_80b_a3b_thinking_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-sonnet-4", | |
| "model_route_id": "anthropic__claude-sonnet-4", | |
| "model_name": "Claude 4 Sonnet 20250514", | |
| "developer": "anthropic", | |
| "variant_key": "20250514", | |
| "raw_model_id": "anthropic/claude-sonnet-4-20250514", | |
| "score": 0.995, | |
| "evaluation_id": "helm_safety/anthropic_claude-sonnet-4-20250514/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-sonnet-4/helm_safety_anthropic_claude_sonnet_4_20250514_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-opus-4", | |
| "model_route_id": "anthropic__claude-opus-4", | |
| "model_name": "Claude 4 Opus 20250514", | |
| "developer": "anthropic", | |
| "variant_key": "20250514", | |
| "raw_model_id": "anthropic/claude-opus-4-20250514", | |
| "score": 0.995, | |
| "evaluation_id": "helm_safety/anthropic_claude-opus-4-20250514/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-opus-4/helm_safety_anthropic_claude_opus_4_20250514_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "xai/grok-3-mini-beta", | |
| "model_route_id": "xai__grok-3-mini-beta", | |
| "model_name": "Grok 3 mini Beta", | |
| "developer": "xai", | |
| "variant_key": "default", | |
| "raw_model_id": "xai/grok-3-mini-beta", | |
| "score": 0.993, | |
| "evaluation_id": "helm_safety/xai_grok-3-mini-beta/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-mini-beta/helm_safety_xai_grok_3_mini_beta_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8", | |
| "model_route_id": "meta__llama-4-maverick-17b-128e-instruct-fp8", | |
| "model_name": "Llama 4 Maverick 17Bx128E Instruct FP8", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-4-maverick-17b-128e-instruct-fp8", | |
| "score": 0.993, | |
| "evaluation_id": "helm_safety/meta_llama-4-maverick-17b-128e-instruct-fp8/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-maverick-17b-128e-instruct-fp8/helm_safety_meta_llama_4_maverick_17b_128e_instruct_fp8_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-3-8b-chat", | |
| "model_route_id": "meta__llama-3-8b-chat", | |
| "model_name": "Llama 3 Instruct 8B", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-3-8b-chat", | |
| "score": 0.993, | |
| "evaluation_id": "helm_safety/meta_llama-3-8b-chat/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-8b-chat/helm_safety_meta_llama_3_8b_chat_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "zai-org/glm-4-5-air-fp8", | |
| "model_route_id": "zai-org__glm-4-5-air-fp8", | |
| "model_name": "GLM-4.5-Air-FP8", | |
| "developer": "zai-org", | |
| "variant_key": "default", | |
| "raw_model_id": "zai-org/glm-4.5-air-fp8", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/zai-org_glm-4.5-air-fp8/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/zai-org__glm-4-5-air-fp8/helm_safety_zai_org_glm_4_5_air_fp8_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen1-5-72b-chat", | |
| "model_route_id": "qwen__qwen1-5-72b-chat", | |
| "model_name": "Qwen1.5 Chat 72B", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen1.5-72b-chat", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/qwen_qwen1.5-72b-chat/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen1-5-72b-chat/helm_safety_qwen_qwen1_5_72b_chat_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/o3-mini", | |
| "model_route_id": "openai__o3-mini", | |
| "model_name": "o3-mini 2025-01-31", | |
| "developer": "openai", | |
| "variant_key": "2025-01-31", | |
| "raw_model_id": "openai/o3-mini-2025-01-31", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/openai_o3-mini-2025-01-31/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3-mini/helm_safety_openai_o3_mini_2025_01_31_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/o3", | |
| "model_route_id": "openai__o3", | |
| "model_name": "o3 2025-04-16", | |
| "developer": "openai", | |
| "variant_key": "2025-04-16", | |
| "raw_model_id": "openai/o3-2025-04-16", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/openai_o3-2025-04-16/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o3/helm_safety_openai_o3_2025_04_16_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/o1", | |
| "model_route_id": "openai__o1", | |
| "model_name": "o1 2024-12-17", | |
| "developer": "openai", | |
| "variant_key": "2024-12-17", | |
| "raw_model_id": "openai/o1-2024-12-17", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/openai_o1-2024-12-17/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1/helm_safety_openai_o1_2024_12_17_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4-turbo", | |
| "model_route_id": "openai__gpt-4-turbo", | |
| "model_name": "GPT-4 Turbo 2024-04-09", | |
| "developer": "openai", | |
| "variant_key": "2024-04-09", | |
| "raw_model_id": "openai/gpt-4-turbo-2024-04-09", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/openai_gpt-4-turbo-2024-04-09/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-turbo/helm_safety_openai_gpt_4_turbo_2024_04_09_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4-1-nano", | |
| "model_route_id": "openai__gpt-4-1-nano", | |
| "model_name": "GPT-4.1 nano 2025-04-14", | |
| "developer": "openai", | |
| "variant_key": "2025-04-14", | |
| "raw_model_id": "openai/gpt-4.1-nano-2025-04-14", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/openai_gpt-4.1-nano-2025-04-14/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4-1-nano/helm_safety_openai_gpt_4_1_nano_2025_04_14_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-3-70b-chat", | |
| "model_route_id": "meta__llama-3-70b-chat", | |
| "model_name": "Llama 3 Instruct 70B", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-3-70b-chat", | |
| "score": 0.99, | |
| "evaluation_id": "helm_safety/meta_llama-3-70b-chat/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-70b-chat/helm_safety_meta_llama_3_70b_chat_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-3-1-8b-instruct-turbo", | |
| "model_route_id": "meta__llama-3-1-8b-instruct-turbo", | |
| "model_name": "Llama 3.1 Instruct Turbo 8B", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-3.1-8b-instruct-turbo", | |
| "score": 0.988, | |
| "evaluation_id": "helm_safety/meta_llama-3.1-8b-instruct-turbo/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-8b-instruct-turbo/helm_safety_meta_llama_3_1_8b_instruct_turbo_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-3-1-405b-instruct-turbo", | |
| "model_route_id": "meta__llama-3-1-405b-instruct-turbo", | |
| "model_name": "Llama 3.1 Instruct Turbo 405B", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-3.1-405b-instruct-turbo", | |
| "score": 0.988, | |
| "evaluation_id": "helm_safety/meta_llama-3.1-405b-instruct-turbo/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-405b-instruct-turbo/helm_safety_meta_llama_3_1_405b_instruct_turbo_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "anthropic/claude-haiku-4-5", | |
| "model_route_id": "anthropic__claude-haiku-4-5", | |
| "model_name": "Claude 4.5 Haiku 20251001", | |
| "developer": "anthropic", | |
| "variant_key": "20251001", | |
| "raw_model_id": "anthropic/claude-haiku-4-5-20251001", | |
| "score": 0.988, | |
| "evaluation_id": "helm_safety/anthropic_claude-haiku-4-5-20251001/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/anthropic__claude-haiku-4-5/helm_safety_anthropic_claude_haiku_4_5_20251001_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "writer/palmyra-med", | |
| "model_route_id": "writer__palmyra-med", | |
| "model_name": "Palmyra Med", | |
| "developer": "writer", | |
| "variant_key": "default", | |
| "raw_model_id": "writer/palmyra-med", | |
| "score": 0.985, | |
| "evaluation_id": "helm_safety/writer_palmyra-med/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/writer__palmyra-med/helm_safety_writer_palmyra_med_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen3-235b-a22b-fp8-tput", | |
| "model_route_id": "qwen__qwen3-235b-a22b-fp8-tput", | |
| "model_name": "Qwen3 235B A22B FP8 Throughput", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen3-235b-a22b-fp8-tput", | |
| "score": 0.985, | |
| "evaluation_id": "helm_safety/qwen_qwen3-235b-a22b-fp8-tput/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen3-235b-a22b-fp8-tput/helm_safety_qwen_qwen3_235b_a22b_fp8_tput_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen2-72b-instruct", | |
| "model_route_id": "qwen__qwen2-72b-instruct", | |
| "model_name": "Qwen2 Instruct 72B", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen2-72b-instruct", | |
| "score": 0.985, | |
| "evaluation_id": "helm_safety/qwen_qwen2-72b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-72b-instruct/helm_safety_qwen_qwen2_72b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4o", | |
| "model_route_id": "openai__gpt-4o", | |
| "model_name": "GPT-4o 2024-05-13", | |
| "developer": "openai", | |
| "variant_key": "2024-05-13", | |
| "raw_model_id": "openai/gpt-4o-2024-05-13", | |
| "score": 0.985, | |
| "evaluation_id": "helm_safety/openai_gpt-4o-2024-05-13/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o/helm_safety_openai_gpt_4o_2024_05_13_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-0-flash-001", | |
| "model_route_id": "google__gemini-2-0-flash-001", | |
| "model_name": "Gemini 2.0 Flash", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.0-flash-001", | |
| "score": 0.985, | |
| "evaluation_id": "helm_safety/google_gemini-2.0-flash-001/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-001/helm_safety_google_gemini_2_0_flash_001_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "deepseek-ai/deepseek-r1-0528", | |
| "model_route_id": "deepseek-ai__deepseek-r1-0528", | |
| "model_name": "DeepSeek-R1-0528", | |
| "developer": "deepseek-ai", | |
| "variant_key": "default", | |
| "raw_model_id": "deepseek-ai/deepseek-r1-0528", | |
| "score": 0.983, | |
| "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-0528/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-0528/helm_safety_deepseek_ai_deepseek_r1_0528_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "ibm/granite-4-0-h-small", | |
| "model_route_id": "ibm__granite-4-0-h-small", | |
| "model_name": "IBM Granite 4.0 Small", | |
| "developer": "ibm", | |
| "variant_key": "default", | |
| "raw_model_id": "ibm/granite-4.0-h-small", | |
| "score": 0.98, | |
| "evaluation_id": "helm_safety/ibm_granite-4.0-h-small/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-h-small/helm_safety_ibm_granite_4_0_h_small_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-5-flash-preview-04-17", | |
| "model_route_id": "google__gemini-2-5-flash-preview-04-17", | |
| "model_name": "Gemini 2.5 Flash 04-17 preview", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.5-flash-preview-04-17", | |
| "score": 0.98, | |
| "evaluation_id": "helm_safety/google_gemini-2.5-flash-preview-04-17/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-preview-04-17/helm_safety_google_gemini_2_5_flash_preview_04_17_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "deepseek-ai/deepseek-r1-hide-reasoning", | |
| "model_route_id": "deepseek-ai__deepseek-r1-hide-reasoning", | |
| "model_name": "DeepSeek R1 hide reasoning", | |
| "developer": "deepseek-ai", | |
| "variant_key": "default", | |
| "raw_model_id": "deepseek-ai/deepseek-r1-hide-reasoning", | |
| "score": 0.98, | |
| "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1-hide-reasoning/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1-hide-reasoning/helm_safety_deepseek_ai_deepseek_r1_hide_reasoning_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "allenai/olmo-2-0325-32b-instruct", | |
| "model_route_id": "allenai__olmo-2-0325-32b-instruct", | |
| "model_name": "OLMo 2 32B Instruct March 2025", | |
| "developer": "allenai", | |
| "variant_key": "default", | |
| "raw_model_id": "allenai/olmo-2-0325-32b-instruct", | |
| "score": 0.98, | |
| "evaluation_id": "helm_safety/allenai_olmo-2-0325-32b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-0325-32b-instruct/helm_safety_allenai_olmo_2_0325_32b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-4o-mini", | |
| "model_route_id": "openai__gpt-4o-mini", | |
| "model_name": "GPT-4o mini 2024-07-18", | |
| "developer": "openai", | |
| "variant_key": "2024-07-18", | |
| "raw_model_id": "openai/gpt-4o-mini-2024-07-18", | |
| "score": 0.978, | |
| "evaluation_id": "helm_safety/openai_gpt-4o-mini-2024-07-18/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-4o-mini/helm_safety_openai_gpt_4o_mini_2024_07_18_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-0-flash-lite-preview-02-05", | |
| "model_route_id": "google__gemini-2-0-flash-lite-preview-02-05", | |
| "model_name": "Gemini 2.0 Flash Lite 02-05 preview", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.0-flash-lite-preview-02-05", | |
| "score": 0.977, | |
| "evaluation_id": "helm_safety/google_gemini-2.0-flash-lite-preview-02-05/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-flash-lite-preview-02-05/helm_safety_google_gemini_2_0_flash_lite_preview_02_05_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "ibm/granite-3-3-8b-instruct", | |
| "model_route_id": "ibm__granite-3-3-8b-instruct", | |
| "model_name": "IBM Granite 3.3 8B Instruct", | |
| "developer": "ibm", | |
| "variant_key": "default", | |
| "raw_model_id": "ibm/granite-3.3-8b-instruct", | |
| "score": 0.975, | |
| "evaluation_id": "helm_safety/ibm_granite-3.3-8b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-3-3-8b-instruct/helm_safety_ibm_granite_3_3_8b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-3-pro-preview", | |
| "model_route_id": "google__gemini-3-pro-preview", | |
| "model_name": "Gemini 3 Pro Preview", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-3-pro-preview", | |
| "score": 0.975, | |
| "evaluation_id": "helm_safety/google_gemini-3-pro-preview/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-3-pro-preview/helm_safety_google_gemini_3_pro_preview_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-0-pro-exp-02-05", | |
| "model_route_id": "google__gemini-2-0-pro-exp-02-05", | |
| "model_name": "Gemini 2.0 Pro 02-05 preview", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.0-pro-exp-02-05", | |
| "score": 0.975, | |
| "evaluation_id": "helm_safety/google_gemini-2.0-pro-exp-02-05/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-0-pro-exp-02-05/helm_safety_google_gemini_2_0_pro_exp_02_05_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-1-5-pro-001", | |
| "model_route_id": "google__gemini-1-5-pro-001", | |
| "model_name": "Gemini 1.5 Pro 001", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-1.5-pro-001", | |
| "score": 0.975, | |
| "evaluation_id": "helm_safety/google_gemini-1.5-pro-001/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-pro-001/helm_safety_google_gemini_1_5_pro_001_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "deepseek-ai/deepseek-r1", | |
| "model_route_id": "deepseek-ai__deepseek-r1", | |
| "model_name": "DeepSeek R1", | |
| "developer": "deepseek-ai", | |
| "variant_key": "default", | |
| "raw_model_id": "deepseek-ai/deepseek-r1", | |
| "score": 0.975, | |
| "evaluation_id": "helm_safety/deepseek-ai_deepseek-r1/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-r1/helm_safety_deepseek_ai_deepseek_r1_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/o1-mini", | |
| "model_route_id": "openai__o1-mini", | |
| "model_name": "o1-mini 2024-09-12", | |
| "developer": "openai", | |
| "variant_key": "2024-09-12", | |
| "raw_model_id": "openai/o1-mini-2024-09-12", | |
| "score": 0.97, | |
| "evaluation_id": "helm_safety/openai_o1-mini-2024-09-12/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__o1-mini/helm_safety_openai_o1_mini_2024_09_12_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-4-scout-17b-16e-instruct", | |
| "model_route_id": "meta__llama-4-scout-17b-16e-instruct", | |
| "model_name": "Llama 4 Scout 17Bx16E Instruct", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-4-scout-17b-16e-instruct", | |
| "score": 0.97, | |
| "evaluation_id": "helm_safety/meta_llama-4-scout-17b-16e-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-4-scout-17b-16e-instruct/helm_safety_meta_llama_4_scout_17b_16e_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-5-pro-preview-03-25", | |
| "model_route_id": "google__gemini-2-5-pro-preview-03-25", | |
| "model_name": "Gemini 2.5 Pro 03-25 preview", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.5-pro-preview-03-25", | |
| "score": 0.97, | |
| "evaluation_id": "helm_safety/google_gemini-2.5-pro-preview-03-25/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-pro-preview-03-25/helm_safety_google_gemini_2_5_pro_preview_03_25_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-1-5-flash-001", | |
| "model_route_id": "google__gemini-1-5-flash-001", | |
| "model_name": "Gemini 1.5 Flash 001", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-1.5-flash-001", | |
| "score": 0.97, | |
| "evaluation_id": "helm_safety/google_gemini-1.5-flash-001/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-1-5-flash-001/helm_safety_google_gemini_1_5_flash_001_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "xai/grok-3-beta", | |
| "model_route_id": "xai__grok-3-beta", | |
| "model_name": "Grok 3 Beta", | |
| "developer": "xai", | |
| "variant_key": "default", | |
| "raw_model_id": "xai/grok-3-beta", | |
| "score": 0.968, | |
| "evaluation_id": "helm_safety/xai_grok-3-beta/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-3-beta/helm_safety_xai_grok_3_beta_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "deepseek-ai/deepseek-llm-67b-chat", | |
| "model_route_id": "deepseek-ai__deepseek-llm-67b-chat", | |
| "model_name": "DeepSeek LLM Chat 67B", | |
| "developer": "deepseek-ai", | |
| "variant_key": "default", | |
| "raw_model_id": "deepseek-ai/deepseek-llm-67b-chat", | |
| "score": 0.968, | |
| "evaluation_id": "helm_safety/deepseek-ai_deepseek-llm-67b-chat/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-llm-67b-chat/helm_safety_deepseek_ai_deepseek_llm_67b_chat_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "google/gemini-2-5-flash-lite", | |
| "model_route_id": "google__gemini-2-5-flash-lite", | |
| "model_name": "Gemini 2.5 Flash-Lite", | |
| "developer": "google", | |
| "variant_key": "default", | |
| "raw_model_id": "google/gemini-2.5-flash-lite", | |
| "score": 0.965, | |
| "evaluation_id": "helm_safety/google_gemini-2.5-flash-lite/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/google__gemini-2-5-flash-lite/helm_safety_google_gemini_2_5_flash_lite_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "qwen/qwen2-5-7b-instruct-turbo", | |
| "model_route_id": "qwen__qwen2-5-7b-instruct-turbo", | |
| "model_name": "Qwen2.5 Instruct Turbo 7B", | |
| "developer": "qwen", | |
| "variant_key": "default", | |
| "raw_model_id": "qwen/qwen2.5-7b-instruct-turbo", | |
| "score": 0.96, | |
| "evaluation_id": "helm_safety/qwen_qwen2.5-7b-instruct-turbo/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/qwen__qwen2-5-7b-instruct-turbo/helm_safety_qwen_qwen2_5_7b_instruct_turbo_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-3-5-turbo-0613", | |
| "model_route_id": "openai__gpt-3-5-turbo-0613", | |
| "model_name": "GPT-3.5 Turbo 0613", | |
| "developer": "openai", | |
| "variant_key": "default", | |
| "raw_model_id": "openai/gpt-3.5-turbo-0613", | |
| "score": 0.958, | |
| "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0613/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0613/helm_safety_openai_gpt_3_5_turbo_0613_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "marin-community/marin-8b-instruct", | |
| "model_route_id": "marin-community__marin-8b-instruct", | |
| "model_name": "Marin 8B Instruct", | |
| "developer": "marin-community", | |
| "variant_key": "default", | |
| "raw_model_id": "marin-community/marin-8b-instruct", | |
| "score": 0.958, | |
| "evaluation_id": "helm_safety/marin-community_marin-8b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/marin-community__marin-8b-instruct/helm_safety_marin_community_marin_8b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "deepseek-ai/deepseek-v3", | |
| "model_route_id": "deepseek-ai__deepseek-v3", | |
| "model_name": "DeepSeek v3", | |
| "developer": "deepseek-ai", | |
| "variant_key": "default", | |
| "raw_model_id": "deepseek-ai/deepseek-v3", | |
| "score": 0.953, | |
| "evaluation_id": "helm_safety/deepseek-ai_deepseek-v3/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/deepseek-ai__deepseek-v3/helm_safety_deepseek_ai_deepseek_v3_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "ibm/granite-4-0-micro", | |
| "model_route_id": "ibm__granite-4-0-micro", | |
| "model_name": "IBM Granite 4.0 Micro", | |
| "developer": "ibm", | |
| "variant_key": "default", | |
| "raw_model_id": "ibm/granite-4.0-micro", | |
| "score": 0.945, | |
| "evaluation_id": "helm_safety/ibm_granite-4.0-micro/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/ibm__granite-4-0-micro/helm_safety_ibm_granite_4_0_micro_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "cohere/command-r", | |
| "model_route_id": "cohere__command-r", | |
| "model_name": "Command R", | |
| "developer": "cohere", | |
| "variant_key": "default", | |
| "raw_model_id": "cohere/command-r", | |
| "score": 0.943, | |
| "evaluation_id": "helm_safety/cohere_command-r/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/cohere__command-r/helm_safety_cohere_command_r_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "mistralai/mistral-small-2501", | |
| "model_route_id": "mistralai__mistral-small-2501", | |
| "model_name": "Mistral Small 3 2501", | |
| "developer": "mistralai", | |
| "variant_key": "default", | |
| "raw_model_id": "mistralai/mistral-small-2501", | |
| "score": 0.932, | |
| "evaluation_id": "helm_safety/mistralai_mistral-small-2501/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-small-2501/helm_safety_mistralai_mistral_small_2501_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "meta/llama-3-1-70b-instruct-turbo", | |
| "model_route_id": "meta__llama-3-1-70b-instruct-turbo", | |
| "model_name": "Llama 3.1 Instruct Turbo 70B", | |
| "developer": "meta", | |
| "variant_key": "default", | |
| "raw_model_id": "meta/llama-3.1-70b-instruct-turbo", | |
| "score": 0.925, | |
| "evaluation_id": "helm_safety/meta_llama-3.1-70b-instruct-turbo/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/meta__llama-3-1-70b-instruct-turbo/helm_safety_meta_llama_3_1_70b_instruct_turbo_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "xai/grok-4-0709", | |
| "model_route_id": "xai__grok-4-0709", | |
| "model_name": "Grok 4 0709", | |
| "developer": "xai", | |
| "variant_key": "default", | |
| "raw_model_id": "xai/grok-4-0709", | |
| "score": 0.922, | |
| "evaluation_id": "helm_safety/xai_grok-4-0709/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/xai__grok-4-0709/helm_safety_xai_grok_4_0709_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-3-5-turbo-1106", | |
| "model_route_id": "openai__gpt-3-5-turbo-1106", | |
| "model_name": "GPT-3.5 Turbo 1106", | |
| "developer": "openai", | |
| "variant_key": "default", | |
| "raw_model_id": "openai/gpt-3.5-turbo-1106", | |
| "score": 0.922, | |
| "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-1106/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-1106/helm_safety_openai_gpt_3_5_turbo_1106_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "openai/gpt-3-5-turbo-0125", | |
| "model_route_id": "openai__gpt-3-5-turbo-0125", | |
| "model_name": "GPT-3.5 Turbo 0125", | |
| "developer": "openai", | |
| "variant_key": "default", | |
| "raw_model_id": "openai/gpt-3.5-turbo-0125", | |
| "score": 0.92, | |
| "evaluation_id": "helm_safety/openai_gpt-3.5-turbo-0125/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/openai__gpt-3-5-turbo-0125/helm_safety_openai_gpt_3_5_turbo_0125_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "mistralai/mixtral-8x22b-instruct-v0-1", | |
| "model_route_id": "mistralai__mixtral-8x22b-instruct-v0-1", | |
| "model_name": "Mixtral Instruct 8x22B", | |
| "developer": "mistralai", | |
| "variant_key": "default", | |
| "raw_model_id": "mistralai/mixtral-8x22b-instruct-v0.1", | |
| "score": 0.915, | |
| "evaluation_id": "helm_safety/mistralai_mixtral-8x22b-instruct-v0.1/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x22b-instruct-v0-1/helm_safety_mistralai_mixtral_8x22b_instruct_v0_1_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "mistralai/mixtral-8x7b-instruct-v0-1", | |
| "model_route_id": "mistralai__mixtral-8x7b-instruct-v0-1", | |
| "model_name": "Mixtral Instruct 8x7B", | |
| "developer": "mistralai", | |
| "variant_key": "default", | |
| "raw_model_id": "mistralai/mixtral-8x7b-instruct-v0.1", | |
| "score": 0.905, | |
| "evaluation_id": "helm_safety/mistralai_mixtral-8x7b-instruct-v0.1/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mixtral-8x7b-instruct-v0-1/helm_safety_mistralai_mixtral_8x7b_instruct_v0_1_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "mistralai/mistral-7b-instruct-v0-3", | |
| "model_route_id": "mistralai__mistral-7b-instruct-v0-3", | |
| "model_name": "Mistral Instruct v0.3 7B", | |
| "developer": "mistralai", | |
| "variant_key": "default", | |
| "raw_model_id": "mistralai/mistral-7b-instruct-v0.3", | |
| "score": 0.81, | |
| "evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.3/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-3/helm_safety_mistralai_mistral_7b_instruct_v0_3_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "allenai/olmo-2-1124-13b-instruct", | |
| "model_route_id": "allenai__olmo-2-1124-13b-instruct", | |
| "model_name": "OLMo 2 13B Instruct November 2024", | |
| "developer": "allenai", | |
| "variant_key": "default", | |
| "raw_model_id": "allenai/olmo-2-1124-13b-instruct", | |
| "score": 0.81, | |
| "evaluation_id": "helm_safety/allenai_olmo-2-1124-13b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-13b-instruct/helm_safety_allenai_olmo_2_1124_13b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "allenai/olmo-2-1124-7b-instruct", | |
| "model_route_id": "allenai__olmo-2-1124-7b-instruct", | |
| "model_name": "OLMo 2 7B Instruct November 2024", | |
| "developer": "allenai", | |
| "variant_key": "default", | |
| "raw_model_id": "allenai/olmo-2-1124-7b-instruct", | |
| "score": 0.775, | |
| "evaluation_id": "helm_safety/allenai_olmo-2-1124-7b-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmo-2-1124-7b-instruct/helm_safety_allenai_olmo_2_1124_7b_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "allenai/olmoe-1b-7b-0125-instruct", | |
| "model_route_id": "allenai__olmoe-1b-7b-0125-instruct", | |
| "model_name": "OLMoE 1B-7B Instruct January 2025", | |
| "developer": "allenai", | |
| "variant_key": "default", | |
| "raw_model_id": "allenai/olmoe-1b-7b-0125-instruct", | |
| "score": 0.725, | |
| "evaluation_id": "helm_safety/allenai_olmoe-1b-7b-0125-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/allenai__olmoe-1b-7b-0125-instruct/helm_safety_allenai_olmoe_1b_7b_0125_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "databricks/dbrx-instruct", | |
| "model_route_id": "databricks__dbrx-instruct", | |
| "model_name": "DBRX Instruct", | |
| "developer": "databricks", | |
| "variant_key": "default", | |
| "raw_model_id": "databricks/dbrx-instruct", | |
| "score": 0.535, | |
| "evaluation_id": "helm_safety/databricks_dbrx-instruct/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/databricks__dbrx-instruct/helm_safety_databricks_dbrx_instruct_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| }, | |
| { | |
| "model_id": "mistralai/mistral-7b-instruct-v0-1", | |
| "model_route_id": "mistralai__mistral-7b-instruct-v0-1", | |
| "model_name": "Mistral Instruct v0.1 7B", | |
| "developer": "mistralai", | |
| "variant_key": "default", | |
| "raw_model_id": "mistralai/mistral-7b-instruct-v0.1", | |
| "score": 0.432, | |
| "evaluation_id": "helm_safety/mistralai_mistral-7b-instruct-v0.1/1777076383.2276576", | |
| "retrieved_timestamp": "1777076383.2276576", | |
| "source_metadata": { | |
| "source_name": "helm_safety", | |
| "source_type": "documentation", | |
| "source_organization_name": "crfm", | |
| "evaluator_relationship": "third_party" | |
| }, | |
| "source_data": { | |
| "dataset_name": "helm_safety", | |
| "source_type": "url", | |
| "url": [ | |
| "https://storage.googleapis.com/crfm-helm-public/safety/benchmark_output/releases/v1.17.0/groups/safety_scenarios.json" | |
| ] | |
| }, | |
| "source_record_url": "https://huggingface.co/datasets/evaleval/card_backend/resolve/main/records/mistralai__mistral-7b-instruct-v0-1/helm_safety_mistralai_mistral_7b_instruct_v0_1_1777076383_2276576.json", | |
| "detailed_evaluation_results": null, | |
| "detailed_evaluation_results_meta": null, | |
| "passthrough_top_level_fields": null, | |
| "instance_level_data": null, | |
| "normalized_result": { | |
| "benchmark_family_key": "helm_safety", | |
| "benchmark_family_name": "SimpleSafetyTests", | |
| "benchmark_parent_key": "helm_safety", | |
| "benchmark_parent_name": "SimpleSafetyTests", | |
| "benchmark_component_key": null, | |
| "benchmark_component_name": null, | |
| "benchmark_leaf_key": "simplesafetytests", | |
| "benchmark_leaf_name": "SimpleSafetyTests", | |
| "slice_key": null, | |
| "slice_name": null, | |
| "metric_name": "LM Evaluated Safety score", | |
| "metric_id": "lm_evaluated_safety_score", | |
| "metric_key": "lm_evaluated_safety_score", | |
| "metric_source": "evaluation_description", | |
| "display_name": "LM Evaluated Safety score", | |
| "canonical_display_name": "SimpleSafetyTests / LM Evaluated Safety score", | |
| "raw_evaluation_name": "SimpleSafetyTests", | |
| "is_summary_score": false | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reproducibility_gap": { | |
| "has_reproducibility_gap": true, | |
| "missing_fields": [ | |
| "temperature", | |
| "max_tokens" | |
| ], | |
| "required_field_count": 2, | |
| "populated_field_count": 0, | |
| "signal_version": "1.0" | |
| }, | |
| "provenance": { | |
| "source_type": "third_party", | |
| "is_multi_source": false, | |
| "first_party_only": false, | |
| "distinct_reporting_organizations": 1, | |
| "signal_version": "1.0" | |
| }, | |
| "variant_divergence": null, | |
| "cross_party_divergence": null | |
| } | |
| } | |
| } | |
| ], | |
| "models_count": 87, | |
| "top_score": 1.0 | |
| } | |
| ], | |
| "subtasks_count": 0, | |
| "metrics_count": 1, | |
| "models_count": 85, | |
| "metric_names": [ | |
| "LM Evaluated Safety score" | |
| ], | |
| "primary_metric_name": "LM Evaluated Safety score", | |
| "top_score": 1.0, | |
| "instance_data": { | |
| "available": false, | |
| "url_count": 0, | |
| "sample_urls": [], | |
| "models_with_loaded_instances": 0 | |
| }, | |
| "evalcards": { | |
| "annotations": { | |
| "reporting_completeness": { | |
| "completeness_score": 0.10714285714285714, | |
| "total_fields_evaluated": 28, | |
| "missing_required_fields": [ | |
| "autobenchmarkcard.benchmark_details.name", | |
| "autobenchmarkcard.benchmark_details.overview", | |
| "autobenchmarkcard.benchmark_details.data_type", | |
| "autobenchmarkcard.benchmark_details.domains", | |
| "autobenchmarkcard.benchmark_details.languages", | |
| "autobenchmarkcard.benchmark_details.similar_benchmarks", | |
| "autobenchmarkcard.benchmark_details.resources", | |
| "autobenchmarkcard.purpose_and_intended_users.goal", | |
| "autobenchmarkcard.purpose_and_intended_users.audience", | |
| "autobenchmarkcard.purpose_and_intended_users.tasks", | |
| "autobenchmarkcard.purpose_and_intended_users.limitations", | |
| "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", | |
| "autobenchmarkcard.methodology.methods", | |
| "autobenchmarkcard.methodology.metrics", | |
| "autobenchmarkcard.methodology.calculation", | |
| "autobenchmarkcard.methodology.interpretation", | |
| "autobenchmarkcard.methodology.baseline_results", | |
| "autobenchmarkcard.methodology.validation", | |
| "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", | |
| "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", | |
| "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", | |
| "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", | |
| "autobenchmarkcard.data", | |
| "evalcards.lifecycle_status", | |
| "evalcards.preregistration_url" | |
| ], | |
| "partial_fields": [], | |
| "field_scores": [ | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.name", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.overview", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.data_type", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.domains", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.languages", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.similar_benchmarks", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.benchmark_details.resources", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.purpose_and_intended_users.goal", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.purpose_and_intended_users.audience", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.purpose_and_intended_users.tasks", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.purpose_and_intended_users.limitations", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.purpose_and_intended_users.out_of_scope_uses", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.methods", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.metrics", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.calculation", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.interpretation", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.baseline_results", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.methodology.validation", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.ethical_and_legal_considerations.privacy_and_anonymity", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.ethical_and_legal_considerations.data_licensing", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.ethical_and_legal_considerations.consent_procedures", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.ethical_and_legal_considerations.compliance_with_regulations", | |
| "coverage_type": "full", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "autobenchmarkcard.data", | |
| "coverage_type": "partial", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "eee_eval.source_metadata.source_type", | |
| "coverage_type": "full", | |
| "score": 1.0 | |
| }, | |
| { | |
| "field_path": "eee_eval.source_metadata.source_organization_name", | |
| "coverage_type": "full", | |
| "score": 1.0 | |
| }, | |
| { | |
| "field_path": "eee_eval.source_metadata.evaluator_relationship", | |
| "coverage_type": "full", | |
| "score": 1.0 | |
| }, | |
| { | |
| "field_path": "evalcards.lifecycle_status", | |
| "coverage_type": "reserved", | |
| "score": 0.0 | |
| }, | |
| { | |
| "field_path": "evalcards.preregistration_url", | |
| "coverage_type": "reserved", | |
| "score": 0.0 | |
| } | |
| ], | |
| "signal_version": "1.0" | |
| }, | |
| "benchmark_comparability": { | |
| "variant_divergence_groups": [], | |
| "cross_party_divergence_groups": [] | |
| } | |
| } | |
| }, | |
| "reproducibility_summary": { | |
| "results_total": 87, | |
| "has_reproducibility_gap_count": 87, | |
| "populated_ratio_avg": 0.0 | |
| }, | |
| "provenance_summary": { | |
| "total_results": 87, | |
| "total_groups": 85, | |
| "multi_source_groups": 0, | |
| "first_party_only_groups": 0, | |
| "source_type_distribution": { | |
| "first_party": 0, | |
| "third_party": 87, | |
| "collaborative": 0, | |
| "unspecified": 0 | |
| } | |
| }, | |
| "comparability_summary": { | |
| "total_groups": 85, | |
| "groups_with_variant_check": 0, | |
| "groups_with_cross_party_check": 0, | |
| "variant_divergent_count": 0, | |
| "cross_party_divergent_count": 0 | |
| } | |
| } |