{ "model": "openai/gpt-oss-120b", "tasks": { "clean_deploy": { "seed_0": { "task": "clean_deploy", "seed_offset": 0, "steps": 15, "reward_sum": -1.48, "rewards_per_step": [ 0.04, 0.01, 0.01, 0.01, -0.15, -0.15, -0.1, -0.1, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15 ], "roles_used": [ "ops", "sre" ], "num_roles_used": 2, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 1.0, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "clean_deploy", "seed_offset": 1, "steps": 15, "reward_sum": -1.0426, "rewards_per_step": [ 0.04, 0.01, 0.01, 0.0277, 0.07, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15, -0.15, 0.1497 ], "roles_used": [ "ops", "sre" ], "num_roles_used": 2, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 1.0, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "clean_deploy", "seed_offset": 2, "steps": 20, "reward_sum": -1.6045, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.0554, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, 0.0275, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } }, "broken_pipeline": { "seed_0": { "task": "broken_pipeline", "seed_offset": 0, "steps": 20, "reward_sum": -1.7022, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0552, 0.0055, -0.05, 0.0058, 0.0058, 0.026, 0.0254, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, -0.05, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "broken_pipeline", "seed_offset": 1, "steps": 20, "reward_sum": -0.2785, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0552, 0.0055, 0.0058, 0.0058, 0.026, 0.0071, 0.0071, 0.0072, 0.007, -0.05, -0.0007, 0.0035, 0.008, -0.05, 0.008, 0.008, -0.05 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "broken_pipeline", "seed_offset": 2, "steps": 20, "reward_sum": -1.6625, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0547, 0.0065, 0.0071, 0.0071, -0.05, 0.0072, 0.007, -0.0007, 0.0035, 0.0275, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } }, "judgment_call": { "seed_0": { "task": "judgment_call", "seed_offset": 0, "steps": 20, "reward_sum": -1.682, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0552, 0.0055, 0.0058, -0.05, 0.0058, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "judgment_call", "seed_offset": 1, "steps": 20, "reward_sum": -1.74, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0547, 0.0065, -0.05, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, -0.05, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "judgment_call", "seed_offset": 2, "steps": 20, "reward_sum": -0.182, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.0065, 0.056, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, 0.008, 0.008, 0.008, -0.05 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } }, "cascading_failure": { "seed_0": { "task": "cascading_failure", "seed_offset": 0, "steps": 20, "reward_sum": -1.6045, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, 0.0483, 0.0035, 0.008, 0.0275, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "cascading_failure", "seed_offset": 1, "steps": 20, "reward_sum": -1.7205, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0552, 0.0055, 0.0058, 0.0058, 0.0065, 0.0071, 0.0266, 0.0072, -0.05, 0.007, -0.0007, -0.05, 0.0035, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "cascading_failure", "seed_offset": 2, "steps": 20, "reward_sum": -1.74, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0544, 0.0058, 0.0058, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, -0.05, 0.008, -0.05, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } }, "capacity_crisis": { "seed_0": { "task": "capacity_crisis", "seed_offset": 0, "steps": 20, "reward_sum": -1.6931, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0253, 0.0241, -0.05, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "capacity_crisis", "seed_offset": 1, "steps": 20, "reward_sum": -1.798, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0552, 0.0055, -0.05, 0.0058, 0.0058, -0.05, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.05, -0.0007, 0.0035, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "capacity_crisis", "seed_offset": 2, "steps": 20, "reward_sum": -0.3274, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.026, 0.0071, 0.0071, 0.0072, 0.007, -0.05, -0.0007, 0.0035, -0.05, 0.008, 0.008, 0.008, -0.05 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } }, "random_incident": { "seed_0": { "task": "random_incident", "seed_offset": 0, "steps": 20, "reward_sum": -1.856, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.0554, 0.0071, 0.0071, 0.0072, -0.05, 0.007, -0.0007, -0.05, 0.0035, -0.05, 0.008, -0.05, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_1": { "task": "random_incident", "seed_offset": 1, "steps": 20, "reward_sum": -0.2889, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0055, 0.0058, 0.0058, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, -0.05, 0.0035, 0.008, 0.008, 0.008, 0.008, -0.05 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 }, "seed_2": { "task": "random_incident", "seed_offset": 2, "steps": 20, "reward_sum": -1.624, "rewards_per_step": [ 0.0432, 0.0048, -0.33, 0.0063, 0.0544, 0.0058, 0.0058, 0.0065, 0.0071, 0.0071, 0.0072, 0.007, -0.0007, 0.0035, 0.008, 0.008, 0.008, 0.008, 0.008, -1.492 ], "roles_used": [ "sre" ], "num_roles_used": 1, "all_3_modes_used": false, "done": true, "success": false, "healthy_ratio": 0.8, "steps_to_recovery": 0, "initial_health": 100.0 } } }, "summary": { "clean_deploy": { "avg_reward": -1.3757, "std_reward": 0.241, "avg_steps": 16.67, "avg_roles_used": 1.67, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 }, "broken_pipeline": { "avg_reward": -1.2144, "std_reward": 0.662, "avg_steps": 20, "avg_roles_used": 1, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 }, "judgment_call": { "avg_reward": -1.2013, "std_reward": 0.7212, "avg_steps": 20, "avg_roles_used": 1, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 }, "cascading_failure": { "avg_reward": -1.6883, "std_reward": 0.0598, "avg_steps": 20, "avg_roles_used": 1, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 }, "capacity_crisis": { "avg_reward": -1.2728, "std_reward": 0.6699, "avg_steps": 20, "avg_roles_used": 1, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 }, "random_incident": { "avg_reward": -1.2563, "std_reward": 0.6906, "avg_steps": 20, "avg_roles_used": 1, "success_rate": 0, "all_3_modes_hit_rate": 0, "avg_steps_to_recovery": 0.0, "recovery_episodes_count": 0, "n": 3 } } }