Spaces:
Sleeping
Sleeping
| """Tests for the reward architecture in training/rewards.py. | |
| Verifies three properties of the reward architecture: | |
| 1. reward_format produces NON-ZERO scores that VARY across rollouts in | |
| a group, so ``reward_std > 0`` and GRPO has a gradient. | |
| 2. The dynamic schedule actually phases format out and phases | |
| environmental in over the planned episode range. | |
| 3. The length monitor trips on sustained drift and stays quiet on | |
| normal traffic. | |
| """ | |
| from __future__ import annotations | |
| from training.rewards import ( | |
| LengthMonitor, | |
| RewardPack, | |
| RewardSchedule, | |
| build_reward_pack, | |
| reward_format, | |
| weighted_environmental_reward, | |
| ) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # reward_format β partial credit + variance across rollouts | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def test_format_perfect_short_output_high(): | |
| perfect = ( | |
| '<thinking>reason</thinking>\n' | |
| '<action id="fs_ls" path="/tmp"/>\n' | |
| '<reversibility level="R1" confidence="0.95"/>' | |
| ) | |
| [score] = reward_format([perfect]) | |
| # 0.20 + 0.15 + 0.15 + 0.10 + 0.10 + 0.10 + 0.20 (brevity <=400) = 1.00 | |
| assert 0.9 <= score <= 1.0 | |
| def test_format_empty_floor_is_nonzero_via_brevity(): | |
| """Empty string gets only the short-length credit.""" | |
| [score] = reward_format([""]) | |
| assert 0.15 <= score <= 0.25 | |
| def test_format_partial_action_only(): | |
| """Action tag present but no reversibility β must earn middle-tier credit.""" | |
| partial = '<action id="fs_ls" path="/tmp"/>' | |
| [score] = reward_format([partial]) | |
| # 0.20 (action) + 0.15 (closed) + 0.20 (short) = 0.55 | |
| assert 0.45 <= score <= 0.65 | |
| def test_format_rambling_is_penalized(): | |
| rambling = "x" * 1200 | |
| [score] = reward_format([rambling]) | |
| # No tags + rambling penalty | |
| assert score <= 0.0 | |
| def test_format_produces_variance_in_a_group(): | |
| """Critical property: a group of diverse rollouts must score differently | |
| so reward_std > 0 in GRPO. was a silent-failure mode when rewards return all zeros.""" | |
| group = [ | |
| "", | |
| '<action id="x"/>', | |
| '<action id="x"/><reversibility level="R3"/>', | |
| '<thinking>x</thinking><action id="x"/><reversibility level="R3" confidence="0.5"/>', | |
| ] | |
| scores = reward_format(group) | |
| distinct = len(set(round(s, 3) for s in scores)) | |
| assert distinct >= 3, f"expected β₯3 distinct rewards, got {distinct}: {scores}" | |
| def test_format_length_tiers_are_monotonic(): | |
| """400 < 600 < 900 < 1100 < rambling β reward must decline as length grows | |
| (holding tag features equal).""" | |
| tags = '<action id="x"/><reversibility level="R1"/>' | |
| scores = reward_format([ | |
| tags, # ~45 chars | |
| tags + "x" * 400, # ~450 | |
| tags + "x" * 700, # ~750 | |
| tags + "x" * 1100, # ~1150 β rambling | |
| ]) | |
| assert scores[0] > scores[1] > scores[2] > scores[3] | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Schedule β format decays, environmental grows | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def test_schedule_format_decays_to_zero(): | |
| s = RewardSchedule(total_episodes=300) | |
| assert s.weight_format(0) == 1.0 | |
| assert s.weight_format(30) < 1.0 | |
| assert s.weight_format(150) == 0.0 | |
| assert s.weight_format(299) == 0.0 | |
| def test_schedule_environmental_grows(): | |
| s = RewardSchedule(total_episodes=300) | |
| assert s.weight_environmental(0) == 0.5 | |
| assert s.weight_environmental(60) > s.weight_environmental(0) | |
| assert s.weight_environmental(150) == 1.5 | |
| assert s.weight_environmental(299) == 1.5 | |
| def test_schedule_weights_sum_is_positive_throughout(): | |
| """At every point in training, total weight must be > 0 so SOMETHING | |
| is being optimized.""" | |
| s = RewardSchedule(total_episodes=300) | |
| for ep in (0, 50, 100, 150, 200, 299): | |
| total = sum(s.weights_at(ep)) | |
| assert total > 0.0, f"Zero total weight at episode {ep}" | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # LengthMonitor β auto-abort behavior | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def test_length_monitor_silent_on_normal_traffic(): | |
| m = LengthMonitor(window=5, threshold_chars=1000, trigger_windows=3) | |
| for _ in range(30): | |
| m.observe("x" * 300) | |
| assert m.abort_flag is False | |
| def test_length_monitor_trips_on_sustained_drift(): | |
| m = LengthMonitor(window=5, threshold_chars=1000, trigger_windows=3) | |
| for _ in range(5): | |
| m.observe("x" * 200) | |
| for _ in range(20): | |
| m.observe("x" * 1200) | |
| assert m.abort_flag is True | |
| def test_length_monitor_tolerates_single_spike(): | |
| """One long completion should not trip the monitor β only sustained drift.""" | |
| m = LengthMonitor(window=5, threshold_chars=1000, trigger_windows=3) | |
| for _ in range(10): | |
| m.observe("x" * 200) | |
| m.observe("x" * 5000) | |
| for _ in range(10): | |
| m.observe("x" * 200) | |
| assert m.abort_flag is False | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # RewardPack composition | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def test_build_reward_pack_has_one_text_func(): | |
| """The text-only pack contains reward_format only; the env reward is | |
| appended separately by stage 3.""" | |
| pack = build_reward_pack(total_episodes=100) | |
| assert len(pack.funcs) == 1 | |
| assert pack.funcs[0].__name__ == "reward_format" | |
| def test_reward_pack_dynamic_weighting(): | |
| pack = build_reward_pack(total_episodes=300) | |
| completion = '<action id="x"/><reversibility level="R1"/>' | |
| pack.episode_counter[0] = 0 | |
| early = pack.funcs[0]([completion])[0] | |
| pack.episode_counter[0] = 200 | |
| late = pack.funcs[0]([completion])[0] | |
| assert early > late | |
| assert late == 0.0 | |
| def test_reward_pack_updates_length_monitor(): | |
| pack = build_reward_pack(total_episodes=100) | |
| long_outputs = ["x" * 1500] * 10 | |
| for _ in range(3): | |
| pack.funcs[0](long_outputs) | |
| assert pack.length_monitor.abort_flag is True | |
| def test_weighted_environmental_reward_applies_schedule(): | |
| """The env reward wrapper must multiply the raw reward by the current | |
| environmental weight.""" | |
| pack = build_reward_pack(total_episodes=300) | |
| def constant_one(completions, **_): | |
| return [1.0] * len(completions) | |
| wrapped = weighted_environmental_reward(constant_one, pack) | |
| pack.episode_counter[0] = 0 | |
| early = wrapped(["x"])[0] | |
| pack.episode_counter[0] = 200 | |
| late = wrapped(["x"])[0] | |
| assert early == 0.5 | |
| assert late == 1.5 | |
| def test_reward_funcs_are_shape_compatible_with_trl(): | |
| """TRL requires reward functions to accept (completions, **kwargs) and | |
| return list[float] the same length as completions.""" | |
| pack = build_reward_pack(total_episodes=100) | |
| completions = [ | |
| '<action id="fs_ls"/><reversibility level="R1" confidence="0.9"/>', | |
| "some bad output", | |
| ] | |
| for fn in pack.funcs: | |
| out = fn( | |
| completions, | |
| actual_r_levels=[1, 4], | |
| task_id=["task_x", "task_y"], | |
| seed=[1, 2], | |
| ) | |
| assert isinstance(out, list) | |
| assert len(out) == len(completions) | |
| assert all(isinstance(x, float) for x in out) | |
| def test_wrappers_survive_trl_keyword_calling_convention(): | |
| """Regression test for a TRL calling-convention bug. | |
| TRL calls reward functions as | |
| ``fn(prompts=[...], completions=[...], task_id=[...], seed=[...])``. | |
| Both wrappers (text pack funcs and the env wrapper) must handle this | |
| without raising "got multiple values for argument 'prompts'".""" | |
| pack = build_reward_pack(total_episodes=100) | |
| completions = ['<action id="fs_ls"/><reversibility level="R1"/>'] | |
| # Text reward β TRL-style keyword call | |
| for fn in pack.funcs: | |
| scores = fn( | |
| prompts=["some prompt"], | |
| completions=completions, | |
| task_id=["task_log_cleanup"], | |
| seed=[0], | |
| ) | |
| assert len(scores) == 1 | |
| # Env wrapper β the function that actually triggered the bug | |
| def fake_env_reward(prompts, completions, **_): | |
| return [0.5] * len(completions) | |
| wrapped = weighted_environmental_reward(fake_env_reward, pack) | |
| scores = wrapped( | |
| prompts=["some prompt"], | |
| completions=completions, | |
| task_id=["task_log_cleanup"], | |
| seed=[0], | |
| ) | |
| assert len(scores) == 1 | |
| assert scores[0] > 0 # schedule weight * 0.5 > 0 | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| # Unlikeliness reward shaping (He et al. 2506.02355) | |
| # βββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββββ | |
| def test_unlikeliness_reward_disabled(): | |
| """BETA_RANK is 0.0 (disabled) because unlikeliness shaping | |
| INVERTED the gradient signal for our classification-style task. Our | |
| continuous partial-credit reward (level_accuracy Γ calibration) meant | |
| top-reward-ranked samples = correct predictions, so the He et al. | |
| penalty on top-ranked samples paid more for WRONG predictions. | |
| With BETA_RANK=0.0, shaped rewards equal raw rewards (times the | |
| schedule weight), so the gradient is clean. | |
| """ | |
| from training.rewards import BETA_RANK | |
| assert BETA_RANK == 0.0, ( | |
| f"Expected BETA_RANK=0.0; got {BETA_RANK}. " | |
| "If you re-enabled unlikeliness shaping, also re-validate that it " | |
| "doesn't invert the gradient for classification-style rewards." | |
| ) | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 # env weight = 1.5 | |
| def raw_returning_spread(completions, **_): | |
| return [1.0, 0.8, 0.6, 0.4] | |
| wrapped = weighted_environmental_reward(raw_returning_spread, pack) | |
| scores = wrapped(completions=["a", "b", "c", "d"]) | |
| # With BETA_RANK=0.0 and no R-level bonus firing (no training_log exposed | |
| # by the raw_fn), the wrapper is just: schedule_weight Γ raw_reward. | |
| # Env weight = 1.5. | |
| assert abs(scores[0] - 1.5 * 1.0) < 1e-6, f"top score wrong: {scores[0]}" | |
| assert abs(scores[3] - 1.5 * 0.4) < 1e-6, f"bottom score wrong: {scores[3]}" | |
| # Ratio of top:bottom preserved (no longer inverted by shaping) | |
| assert abs(scores[0] / scores[3] - 1.0 / 0.4) < 1e-6 | |
| def test_unlikeliness_reward_passes_negatives_unchanged(): | |
| """With BETA_RANK=0.0, negative rewards flow through unchanged too | |
| (previously shaping only affected positives; now nothing is shaped).""" | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 | |
| def raw(completions, **_): | |
| return [0.8, -0.1, -0.1, -0.1] | |
| wrapped = weighted_environmental_reward(raw, pack) | |
| scores = wrapped(completions=["a", "b", "c", "d"]) | |
| # No penalty on top (BETA_RANK=0.0) | |
| assert abs(scores[0] - 1.5 * 0.8) < 1e-6, f"top shouldn't be penalized now: {scores[0]}" | |
| # Negatives still flow through | |
| for s in scores[1:]: | |
| assert abs(s - 1.5 * -0.1) < 1e-6, f"negative reward shaped unexpectedly: {s}" | |
| def test_r_level_bonus_applied_for_correct_high_r_predictions(): | |
| """When the raw_fn exposes a training_log and the last G entries show | |
| correctly-predicted R4 or R5 actions, a bonus is added before the | |
| schedule weight multiplies. This directly incentivizes developing | |
| the R4/R5 prediction capability on classes the policy underweights.""" | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 # env weight = 1.5 | |
| # Build a fake raw_fn with a training_log attribute (matching | |
| # _make_task_reward's contract in stage_3_grpo) | |
| training_log = [ | |
| {"predicted_r_level": 5, "actual_r_level": 5}, # correct R5 β +0.2 | |
| {"predicted_r_level": 4, "actual_r_level": 4}, # correct R4 β +0.1 | |
| ] | |
| def raw(completions, **_): | |
| return [0.5, 0.5] | |
| raw.training_log = training_log | |
| wrapped = weighted_environmental_reward(raw, pack) | |
| scores = wrapped(completions=["a", "b"]) | |
| # Without shaping: both are 0.5. With unlikeliness (2 samples, rank 0 and | |
| # rank 1 normalized are 1/2=0.5 and 0): sorted descending [0.5, 0.5] β | |
| # both same, arbitrary ranking. Since rewards are identical, the rank | |
| # order is stable but the penalty is asymmetric. The key test is: the | |
| # R-level bonus actually fires and changes the final scores compared | |
| # to no-bonus baseline. | |
| def raw_no_bonus(completions, **_): | |
| return [0.5, 0.5] | |
| wrapped_no_bonus = weighted_environmental_reward(raw_no_bonus, pack) | |
| baseline = wrapped_no_bonus(completions=["a", "b"]) | |
| # Bonus fires for both entries; shaped reward must be > baseline | |
| assert scores[0] > baseline[0], f"R5 bonus did not fire: {scores[0]} vs baseline {baseline[0]}" | |
| assert scores[1] > baseline[1], f"R4 bonus did not fire: {scores[1]} vs baseline {baseline[1]}" | |
| def test_r_level_bonus_skipped_for_wrong_predictions(): | |
| """If predicted != actual, no bonus.""" | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 | |
| training_log = [ | |
| {"predicted_r_level": 2, "actual_r_level": 5}, # wrong, no bonus | |
| ] | |
| def raw(completions, **_): | |
| return [0.5] | |
| raw.training_log = training_log | |
| wrapped = weighted_environmental_reward(raw, pack) | |
| [score] = wrapped(completions=["a"]) | |
| # Only 1 sample β no rank shaping, no bonus. Just schedule weight. | |
| expected = 1.5 * 0.5 | |
| assert abs(score - expected) < 1e-6, f"wrong prediction got bonus: {score} vs {expected}" | |
| def test_r_level_bonus_skipped_for_low_r_predictions(): | |
| """R1/R2/R3 predictions get no bonus even when correct β only the | |
| rare high-R levels (R4, R5) incentivize the policy to develop them.""" | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 | |
| training_log = [ | |
| {"predicted_r_level": 2, "actual_r_level": 2}, # correct R2, no bonus | |
| {"predicted_r_level": 1, "actual_r_level": 1}, # correct R1, no bonus | |
| ] | |
| def raw(completions, **_): | |
| return [0.5, 0.5] | |
| raw.training_log = training_log | |
| wrapped = weighted_environmental_reward(raw, pack) | |
| scores = wrapped(completions=["a", "b"]) | |
| # No R-level bonus fired. Only schedule weight + unlikeliness (which is | |
| # symmetric for identical rewards). The key check: nothing above the | |
| # expected shaped value. | |
| # With 2 samples and equal raw 0.5, sorted desc: indices could go either | |
| # way but rank 0 gets 0.5*(1-0.25*1.0)=0.375 and rank 1 gets | |
| # 0.5*(1-0.25*0)=0.5. So after scheduling (Γ1.5): scores are {0.5625, 0.75}. | |
| # Both scores must be bounded above by 1.5*0.5=0.75. | |
| for s in scores: | |
| assert s <= 1.5 * 0.5 + 1e-6, f"low-R prediction got unexpected bonus: {s}" | |
| def test_r_level_bonus_scales_with_r_level(): | |
| """The bonus scales R_LEVEL_BONUS_PER_LEVEL Γ (actual_r_level - 3), so | |
| R5 yields 2Γ the R4 bonus. This rewards the model more for developing | |
| the rarest, most valuable prediction capability.""" | |
| from training.rewards import R_LEVEL_BONUS_PER_LEVEL | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 200 | |
| # One-sample groups, so no unlikeliness shaping interferes | |
| training_log_r4 = [{"predicted_r_level": 4, "actual_r_level": 4}] | |
| def raw_r4(completions, **_): | |
| return [0.0] | |
| raw_r4.training_log = training_log_r4 | |
| wrapped_r4 = weighted_environmental_reward(raw_r4, pack) | |
| [r4_score] = wrapped_r4(completions=["a"]) | |
| training_log_r5 = [{"predicted_r_level": 5, "actual_r_level": 5}] | |
| def raw_r5(completions, **_): | |
| return [0.0] | |
| raw_r5.training_log = training_log_r5 | |
| wrapped_r5 = weighted_environmental_reward(raw_r5, pack) | |
| [r5_score] = wrapped_r5(completions=["a"]) | |
| # R5 bonus = 0.1 * 2 = 0.2. R4 bonus = 0.1 * 1 = 0.1. Schedule weight 1.5. | |
| assert abs(r4_score - 1.5 * R_LEVEL_BONUS_PER_LEVEL) < 1e-6, f"R4 bonus wrong: {r4_score}" | |
| assert abs(r5_score - 1.5 * R_LEVEL_BONUS_PER_LEVEL * 2) < 1e-6, f"R5 bonus wrong: {r5_score}" | |
| assert r5_score > r4_score, "R5 bonus should exceed R4" | |
| def test_wrapper_is_robust_to_missing_training_log(): | |
| """If raw_fn doesn't expose training_log (e.g. test fakes), the wrapper | |
| must not crash β it just skips the R-level bonus step.""" | |
| pack = build_reward_pack(total_episodes=300) | |
| pack.episode_counter[0] = 100 | |
| def raw_no_log(completions, **_): | |
| return [0.5, 0.5] | |
| # No training_log attribute at all | |
| wrapped = weighted_environmental_reward(raw_no_log, pack) | |
| scores = wrapped(completions=["a", "b"]) | |
| assert len(scores) == 2 | |
| assert all(s > 0 for s in scores) | |