Transformers
Safetensors
trl
grpo
arabic-poetry
classical-arabic
lora
AhmadAbbass commited on
Commit
61c9ac0
·
verified ·
1 Parent(s): d487cc2

Training in progress, step 1200

Browse files
adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:72ebcb8e2a1122734305b533826d37fd058a5ac12a0d1ba763ebc116384913d3
3
  size 639691872
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:13011eb73955c3206452399fdbe91b465de038ded1230e4da5acc410aa824763
3
  size 639691872
all_generations.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:359d528960a35ec6991882f101680748220558f114eaca752b65a8d810dd86fa
3
- size 208776691
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9dd80f1c2380453db6d5b55925515e83f52cbd78c6df90e9a139d242c71e4a84
3
+ size 217398123
checkpoint_events.jsonl CHANGED
@@ -44,3 +44,5 @@
44
  {"timestamp_utc": "2026-04-11T21:45:19Z", "event_type": "checkpoint_saved", "global_step": 1100, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/checkpoint-1100", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
45
  {"timestamp_utc": "2026-04-11T21:51:32Z", "event_type": "evaluation_completed", "global_step": 1150, "metrics": {"eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154}}
46
  {"timestamp_utc": "2026-04-11T21:51:35Z", "event_type": "checkpoint_saved", "global_step": 1150, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/checkpoint-1150", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
 
 
 
44
  {"timestamp_utc": "2026-04-11T21:45:19Z", "event_type": "checkpoint_saved", "global_step": 1100, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/checkpoint-1100", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
45
  {"timestamp_utc": "2026-04-11T21:51:32Z", "event_type": "evaluation_completed", "global_step": 1150, "metrics": {"eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154}}
46
  {"timestamp_utc": "2026-04-11T21:51:35Z", "event_type": "checkpoint_saved", "global_step": 1150, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/checkpoint-1150", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
47
+ {"timestamp_utc": "2026-04-11T21:57:39Z", "event_type": "evaluation_completed", "global_step": 1200, "metrics": {"eval_loss": NaN, "eval_runtime": 90.6551, "eval_samples_per_second": 1.147, "eval_steps_per_second": 0.143}}
48
+ {"timestamp_utc": "2026-04-11T21:57:42Z", "event_type": "checkpoint_saved", "global_step": 1200, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/checkpoint-1200", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
metrics.csv CHANGED
@@ -1173,3 +1173,54 @@ clip_ratio/high_max,clip_ratio/high_mean,clip_ratio/low_mean,clip_ratio/low_min,
1173
  0.009015594609081745,0.009015594609081745,0.004067460540682077,0.004067460540682077,0.013083055149763823,0.0,63.0,63.0,57.5,57.5,54.0,54.0,0.05149026960134506,0.04440840284213778,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1150,5.2525129318237305,6.5181818181818195e-06,0.0522,train,2483289.0,0.8777114748954773,1.0,0.0,1.0,0.0,0.8777114748954773,0.16049861907958984,0.16049860417842865,0.8777114748954773,0.16049861907958984,0.8777114748954773,1.0,0.0,1.0,0.0,0.8777114748954773,0.16049861907958984,0.8777114748954773,0.16049861907958984,1.5273561477661133,0.9977035522460938,0.006705376319587231,5.00484561920166,0.024218173697590828,2026-04-11T21:50:08Z
1174
  ,,,,,,,,,,,,,0.04440840284213778,0.0,0.0,0.0,0.0,0.0,0.057692307692307696,447.0,396.38461538461536,216.47115384615384,197.60806157038763,60.76923076923077,60.76923076923077,0.017647026679836787,0.0,nan,2483289.0,0.511953374514213,1.0,0.0,0.9469390053015488,0.06932907207654072,0.5331913347427661,0.45816060442190903,nan,0.511953374514213,0.440208015533594,0.511953374514213,1.0,0.0,0.9469390053015488,0.06932907207654072,0.5331913347427661,0.45816060442190903,0.511953374514213,0.440208015533594,84.6125,1.229,1.403554081916809,1.0007386207580566,0.629830559858909,0.5305701494216919,0.0024267814587801695,0.154,,1150,,,,eval,,,,,,,,,,,,,,,,,,,,,,,,,,2026-04-11T21:51:32Z
1175
  0.006250000325962901,0.006250000325962901,0.003968254197388887,0.003968254197388887,0.010218254523351789,0.0,136.0,136.0,122.75,122.75,120.0,120.0,0.013336000498384237,0.04444701884460921,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1151,3.7881219387054443,6.515151515151516e-06,0.0504,train,2485639.0,0.888770341873169,1.0,0.0,1.0,0.0,0.888770341873169,0.21289391815662384,0.21289391815662384,0.888770341873169,0.21289391815662384,0.888770341873169,1.0,0.0,1.0,0.0,0.888770341873169,0.21289391815662384,0.888770341873169,0.21289391815662384,1.9864379167556763,0.9987115263938904,0.0021391669288277626,6.1473388671875,0.020240498706698418,2026-04-11T21:51:41Z
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1173
  0.009015594609081745,0.009015594609081745,0.004067460540682077,0.004067460540682077,0.013083055149763823,0.0,63.0,63.0,57.5,57.5,54.0,54.0,0.05149026960134506,0.04440840284213778,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1150,5.2525129318237305,6.5181818181818195e-06,0.0522,train,2483289.0,0.8777114748954773,1.0,0.0,1.0,0.0,0.8777114748954773,0.16049861907958984,0.16049860417842865,0.8777114748954773,0.16049861907958984,0.8777114748954773,1.0,0.0,1.0,0.0,0.8777114748954773,0.16049861907958984,0.8777114748954773,0.16049861907958984,1.5273561477661133,0.9977035522460938,0.006705376319587231,5.00484561920166,0.024218173697590828,2026-04-11T21:50:08Z
1174
  ,,,,,,,,,,,,,0.04440840284213778,0.0,0.0,0.0,0.0,0.0,0.057692307692307696,447.0,396.38461538461536,216.47115384615384,197.60806157038763,60.76923076923077,60.76923076923077,0.017647026679836787,0.0,nan,2483289.0,0.511953374514213,1.0,0.0,0.9469390053015488,0.06932907207654072,0.5331913347427661,0.45816060442190903,nan,0.511953374514213,0.440208015533594,0.511953374514213,1.0,0.0,0.9469390053015488,0.06932907207654072,0.5331913347427661,0.45816060442190903,0.511953374514213,0.440208015533594,84.6125,1.229,1.403554081916809,1.0007386207580566,0.629830559858909,0.5305701494216919,0.0024267814587801695,0.154,,1150,,,,eval,,,,,,,,,,,,,,,,,,,,,,,,,,2026-04-11T21:51:32Z
1175
  0.006250000325962901,0.006250000325962901,0.003968254197388887,0.003968254197388887,0.010218254523351789,0.0,136.0,136.0,122.75,122.75,120.0,120.0,0.013336000498384237,0.04444701884460921,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1151,3.7881219387054443,6.515151515151516e-06,0.0504,train,2485639.0,0.888770341873169,1.0,0.0,1.0,0.0,0.888770341873169,0.21289391815662384,0.21289391815662384,0.888770341873169,0.21289391815662384,0.888770341873169,1.0,0.0,1.0,0.0,0.888770341873169,0.21289391815662384,0.888770341873169,0.21289391815662384,1.9864379167556763,0.9987115263938904,0.0021391669288277626,6.1473388671875,0.020240498706698418,2026-04-11T21:51:41Z
1176
+ 0.0,0.0,0.0,0.0,0.0,0.0,61.0,61.0,61.0,61.0,61.0,61.0,0.0009597428434062749,0.04448563484708063,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1152,0.0,6.512121212121213e-06,0.0,train,2487423.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.002875804901123,1.0000879764556885,0.9996728897094727,0.002871689386665821,9.262973617296666e-05,2026-04-11T21:51:45Z
1177
+ 0.0,0.0,0.0,0.0,0.0,0.0,66.0,66.0,66.0,66.0,66.0,66.0,0.00251059714355506,0.044524250849552055,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1153,0.0,6.5090909090909095e-06,0.0,train,2489239.0,0.9972342252731323,1.0,0.0,1.0,0.0,0.9972342252731323,0.0,0.0,0.9972342252731323,0.0,0.9972342252731323,1.0,0.0,1.0,0.0,0.9972342252731323,0.0,0.9972342252731323,0.0,1.0045279264450073,1.0000462532043457,0.9860818386077881,0.014015945605933666,0.00015291250019799918,2026-04-11T21:51:50Z
1178
+ 0.008954269345849752,0.008954269345849752,0.013047155574895442,0.013047155574895442,0.022001424920745194,0.0,71.0,71.0,65.5,65.5,53.0,53.0,0.0860102130100131,0.04456286685202348,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1154,5.494170665740967,6.506060606060607e-06,0.096,train,2491171.0,0.13582952320575714,1.0,0.0,1.0,0.0,0.13582952320575714,0.19803810119628906,0.19803808629512787,0.13582952320575714,0.19803810119628906,0.13582952320575714,1.0,0.0,1.0,0.0,0.13582952320575714,0.19803810119628906,0.13582952320575714,0.19803810119628906,1.913017988204956,0.9954270124435425,0.015514720231294632,4.165966033935547,0.041303541511297226,2026-04-11T21:51:55Z
1179
+ 0.0,0.0,0.0,0.0,0.0,0.0,130.0,130.0,130.0,130.0,130.0,130.0,0.0005436375031422358,0.0446014828544949,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1155,0.0,6.503030303030303e-06,0.0,train,2493723.0,0.9970986843109131,1.0,0.0,1.0,0.0,0.9970986843109131,0.0,0.0,0.9970986843109131,0.0,0.9970986843109131,1.0,0.0,1.0,0.0,0.9970986843109131,0.0,0.9970986843109131,0.0,1.0045697689056396,0.9999405145645142,0.8893400430679321,0.117275670170784,0.0001729379582684487,2026-04-11T21:52:02Z
1180
+ 0.001953125,0.001953125,0.014450150192715228,0.014450150192715228,0.016403275192715228,0.125,512.0,288.0,304.5,274.8571472167969,256.0,256.0,0.04777472233399749,0.04464009885696633,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1156,3.3614652156829834,6.5000000000000004e-06,-0.0346,train,2497463.0,0.025792036205530167,0.875,0.3535533845424652,0.9285714626312256,0.2020305097103119,0.028450578451156616,0.04791045933961868,0.048944856971502304,0.025792036205530167,0.048944856971502304,0.025792036205530167,0.875,0.3535533845424652,0.9285714626312256,0.2020305097103119,0.028450578451156616,0.04791045933961868,0.025792036205530167,0.048944856971502304,2.0,1.001096487045288,0.015113191679120064,4.192187309265137,0.024456597864627838,2026-04-11T21:52:12Z
1181
+ 0.0,0.0,0.0,0.0,0.0,0.0,66.0,66.0,66.0,66.0,66.0,66.0,0.006829831196228042,0.04467871485943775,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1157,0.0,6.496969696969697e-06,0.0,train,2499215.0,0.9972342252731323,1.0,0.0,1.0,0.0,0.9972342252731323,0.0,0.0,0.9972342252731323,0.0,0.9972342252731323,1.0,0.0,1.0,0.0,0.9972342252731323,0.0,0.9972342252731323,0.0,1.0112634897232056,0.9938147664070129,0.03500813990831375,3.352174758911133,0.0148831931874156,2026-04-11T21:52:16Z
1182
+ 0.0,0.0,0.0,0.0,0.0,0.0,72.0,72.0,72.0,72.0,72.0,72.0,0.0018330513557884842,0.044717330861909176,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1158,0.0,6.493939393939395e-06,0.0,train,2501127.0,0.9888595938682556,1.0,0.0,1.0,0.0,0.9888595938682556,0.0,0.0,0.9888595938682556,0.0,0.9888595938682556,1.0,0.0,1.0,0.0,0.9888595938682556,0.0,0.9888595938682556,0.0,1.0043905973434448,1.000144600868225,0.9968689680099487,0.0043810089118778706,0.00017435323388781399,2026-04-11T21:52:21Z
1183
+ 0.0,0.0,0.0,0.0,0.0,0.0,91.0,91.0,91.0,91.0,91.0,91.0,0.0013790328812319785,0.0447559468643806,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1159,0.0,6.490909090909091e-06,0.0,train,2503135.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.0065501928329468,1.0001311302185059,0.9994356036186218,0.006528853438794613,0.00013290751667227596,2026-04-11T21:52:26Z
1184
+ 0.0,0.0,0.0,0.0,0.0,0.0,62.0,62.0,62.0,62.0,62.0,62.0,0.001158960752945859,0.044794562866852024,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1160,0.0,6.487878787878789e-06,0.0,train,2504759.0,0.9959487915039062,1.0,0.0,1.0,0.0,0.9959487915039062,0.0,0.0,0.9959487915039062,0.0,0.9959487915039062,1.0,0.0,1.0,0.0,0.9959487915039062,0.0,0.9959487915039062,0.0,1.005993127822876,1.0000890493392944,0.9942424893379211,0.005975149571895599,0.0001226484455401078,2026-04-11T21:52:31Z
1185
+ 0.0,0.0,0.0,0.0,0.0,0.0,96.0,96.0,96.0,96.0,96.0,96.0,0.0013596047574537806,0.04483317886932345,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1161,0.0,6.484848484848485e-06,0.0,train,2506847.0,0.9888964295387268,1.0,0.0,1.0,0.0,0.9888964295387268,0.0,0.0,0.9888964295387268,0.0,0.9888964295387268,1.0,0.0,1.0,0.0,0.9888964295387268,0.0,0.9888964295387268,0.0,1.0041718482971191,1.0000895261764526,0.9944958090782166,0.005519423168152571,0.00014731692499481142,2026-04-11T21:52:36Z
1186
+ 0.004590633558109403,0.004590633558109403,0.004629629664123058,0.004629629664123058,0.009220263222232461,0.0,56.0,56.0,54.375,54.375,53.0,53.0,0.028170868754386902,0.04487179487179487,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1162,9.667710304260254,6.481818181818182e-06,-0.0067,train,2508562.0,0.9774544835090637,1.0,0.0,1.0,0.0,0.9774544835090637,0.007204634603112936,0.007204628549516201,0.9774544835090637,0.007204634603112936,0.9774544835090637,1.0,0.0,1.0,0.0,0.9774544835090637,0.007204634603112936,0.9774544835090637,0.007204634603112936,2.0,0.9978480339050293,0.02039913274347782,3.8922629356384277,0.024649647995829582,2026-04-11T21:52:41Z
1187
+ 0.0,0.0,0.0008620689623057842,0.0008620689623057842,0.0008620689623057842,0.0,145.0,145.0,145.0,145.0,145.0,145.0,0.006222540338058025,0.044910410874266296,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1163,0.0491514727473259,6.478787878787879e-06,-0.0003,train,2511042.0,0.9984663724899292,1.0,0.0,1.0,0.0,0.9984663724899292,3.918135462299688e-06,3.918005404557334e-06,0.9984663724899292,3.918135462299688e-06,0.9984663724899292,1.0,0.0,1.0,0.0,0.9984663724899292,3.918135462299688e-06,0.9984663724899292,3.918135462299688e-06,1.9562904834747314,1.000813603401184,0.8294064402580261,0.6710500717163086,0.0012445234460756183,2026-04-11T21:52:47Z
1188
+ 0.0,0.0,0.0,0.0,0.0,0.0,61.0,61.0,61.0,61.0,61.0,61.0,0.0006748539672116749,0.04494902687673772,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1164,0.0,6.475757575757576e-06,0.0,train,2512906.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.0016183853149414,1.000075340270996,0.9988471865653992,0.0016170135932043195,8.520409755874425e-05,2026-04-11T21:52:51Z
1189
+ 0.0,0.0,0.0,0.0,0.0,0.0,62.0,62.0,62.0,62.0,62.0,62.0,0.0014549151237588376,0.044987642879209144,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1165,0.0,6.472727272727272e-06,0.0,train,2514546.0,0.9959487915039062,1.0,0.0,1.0,0.0,0.9959487915039062,0.0,0.0,0.9959487915039062,0.0,0.9959487915039062,1.0,0.0,1.0,0.0,0.9959487915039062,0.0,0.9959487915039062,0.0,1.003846287727356,1.0001657009124756,0.996885359287262,0.003838915377855301,0.00018705571710597724,2026-04-11T21:52:56Z
1190
+ 0.008513708598911762,0.008513708598911762,0.006657268386334181,0.006657268386334181,0.015170976985245943,0.0,63.0,63.0,58.0,58.0,55.0,55.0,0.059585667215287685,0.04502625888168057,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1166,190.9034881591797,6.4696969696969705e-06,0.0454,train,2516434.0,0.996626615524292,1.0,0.0,1.0,0.0,0.996626615524292,0.001510909991338849,0.0015109025407582521,0.996626615524292,0.001510909991338849,0.996626615524292,1.0,0.0,1.0,0.0,0.996626615524292,0.001510909991338849,0.996626615524292,0.001510909991338849,1.7860766649246216,0.9995715618133545,0.3032132685184479,1.1933188438415527,0.025698505342006683,2026-04-11T21:53:01Z
1191
+ 0.0,0.0,0.0,0.0,0.0,0.0,48.0,48.0,48.0,48.0,48.0,48.0,0.006112938834121451,0.04506487488415199,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1167,0.0,6.466666666666667e-06,0.0,train,2518002.0,0.9887858629226685,1.0,0.0,1.0,0.0,0.9887858629226685,0.0,0.0,0.9887858629226685,0.0,0.9887858629226685,1.0,0.0,1.0,0.0,0.9887858629226685,0.0,0.9887858629226685,0.0,1.032031774520874,0.9993685483932495,0.8270683288574219,0.18986795842647552,0.0013065863167867064,2026-04-11T21:53:06Z
1192
+ 0.004611280397512019,0.004611280397512019,0.014004629920236766,0.014004629920236766,0.018615910317748785,0.0,91.0,91.0,81.875,81.875,80.0,80.0,0.03918551583774388,0.04510349088662342,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1168,7.1253252029418945,6.463636363636364e-06,0.0287,train,2520049.0,0.038292575627565384,1.0,0.0,0.9583333730697632,0.117851123213768,0.03959798067808151,0.011728010140359402,0.01326083205640316,0.038292575627565384,0.01326083205640316,0.038292575627565384,1.0,0.0,0.9583333730697632,0.117851123213768,0.03959798067808151,0.011728010140359402,0.038292575627565384,0.01326083205640316,1.6731361150741577,0.9953240752220154,0.008236486464738846,4.7991814613342285,0.028142500668764114,2026-04-11T21:53:11Z
1193
+ 0.002016128972172737,0.002016128972172737,0.006818181602284312,0.006818181602284312,0.00883431057445705,0.0,63.0,63.0,58.75,58.75,55.0,55.0,0.0419846111908555,0.04514210688909484,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1169,4.53950834274292,6.460606060606061e-06,-0.044,train,2521727.0,0.9976530075073242,1.0,0.0,1.0,0.0,0.9976530075073242,0.0003900358860846609,0.00039003457641229033,0.9976530075073242,0.0003900358860846609,0.9976530075073242,1.0,0.0,1.0,0.0,0.9976530075073242,0.0003900358860846609,0.9976530075073242,0.0003900358860846609,1.4813544750213623,0.9979230165481567,0.4586032032966614,0.7795699834823608,0.013675752095878124,2026-04-11T21:53:15Z
1194
+ 0.0,0.0,0.0,0.0,0.0,1.0,512.0,0.0,512.0,0.0,512.0,0.0,0.0,0.045180722891566265,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1170,0.0,6.457575757575758e-06,0.0,train,2523215.0,0.8545387387275696,1.0,0.0,0.8571428656578064,0.0,0.9969618916511536,0.0,0.0,0.8545387387275696,0.0,0.8545387387275696,1.0,0.0,0.8571428656578064,0.0,0.9969618916511536,0.0,0.8545387387275696,0.0,0.0,0.0,0.0,0.0,0.0,2026-04-11T21:53:26Z
1195
+ 0.0,0.0,0.0,0.0,0.0,0.0,61.0,61.0,61.0,61.0,61.0,61.0,0.0013301519793458283,0.04521933889403769,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1171,0.0,6.454545454545456e-06,0.0,train,2524991.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.0045820474624634,1.0001263618469238,0.999534547328949,0.004571585915982723,0.00013418152229860425,2026-04-11T21:53:30Z
1196
+ 0.0,0.0,0.0,0.0,0.0,0.0,96.0,96.0,96.0,96.0,96.0,96.0,0.00159282027016161,0.04525795489650911,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1172,0.0,6.451515151515152e-06,0.0,train,2527167.0,0.9888964295387268,1.0,0.0,1.0,0.0,0.9888964295387268,0.0,0.0,0.9888964295387268,0.0,0.9888964295387268,1.0,0.0,1.0,0.0,0.9888964295387268,0.0,0.9888964295387268,0.0,1.0190335512161255,1.00016450881958,0.9549573659896851,0.04608858376741409,0.0002919211983680725,2026-04-11T21:53:36Z
1197
+ 0.0026041667442768812,0.0026041667442768812,0.004807692486792803,0.004807692486792803,0.007411859231069684,0.0,52.0,52.0,48.5,48.5,48.0,48.0,0.018341065326239914,0.04529657089898054,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1173,20.588041305541992,6.4484848484848496e-06,0.0337,train,2528827.0,0.9464411735534668,1.0,0.0,1.0,0.0,0.9464411735534668,0.11976895481348038,0.11976895481348038,0.9464411735534668,0.11976895481348038,0.9464411735534668,1.0,0.0,1.0,0.0,0.9464411735534668,0.11976895481348038,0.9464411735534668,0.11976895481348038,1.9219340085983276,1.0024529695510864,0.244814932346344,1.4072527885437012,0.008157458156347275,2026-04-11T21:53:41Z
1198
+ 0.004166666883975267,0.004166666883975267,0.0,0.0,0.004166666883975267,0.0,33.0,33.0,30.625,30.625,30.0,30.0,0.015783087466843426,0.04533518690145196,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1174,12.414958953857422,6.445454545454546e-06,0.032,train,2530280.0,0.9770581126213074,1.0,0.0,1.0,0.0,0.9770581126213074,0.03885127976536751,0.03885127976536751,0.9770581126213074,0.03885127976536751,0.9770581126213074,1.0,0.0,1.0,0.0,0.9770581126213074,0.03885127976536751,0.9770581126213074,0.03885127976536751,1.4162566661834717,1.0016231536865234,0.7204657793045044,0.34801721572875977,0.00510073360055685,2026-04-11T21:53:45Z
1199
+ 0.0059843831695616245,0.0059843831695616245,0.001953125,0.001953125,0.007937508169561625,0.0,64.0,64.0,62.75,62.75,62.0,62.0,0.03662474290467799,0.045373802903923385,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1175,10.53234577178955,6.442424242424243e-06,0.0157,train,2531910.0,0.98736572265625,1.0,0.0,1.0,0.0,0.98736572265625,0.02987661585211754,0.02987661026418209,0.98736572265625,0.02987661585211754,0.98736572265625,1.0,0.0,1.0,0.0,0.98736572265625,0.02987661585211754,0.98736572265625,0.02987661585211754,1.5600427389144897,1.0028626918792725,0.36092060804367065,1.0190973281860352,0.009396832436323166,2026-04-11T21:53:50Z
1200
+ 0.0,0.0,0.0004071661096531898,0.0004071661096531898,0.0004071661096531898,0.0,307.0,307.0,307.0,307.0,307.0,307.0,0.001436470149201341,0.04541241890639481,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1176,0.008205018006265163,6.43939393939394e-06,0.0004,train,2536102.0,0.8875229358673096,1.0,0.0,0.8888888955116272,0.0,0.9984632730484009,1.9595856883825036e-06,1.7291217773163226e-06,0.8875229358673096,1.7389269260092988e-06,0.8875229358673096,1.0,0.0,0.8888888955116272,0.0,0.9984632730484009,1.9595856883825036e-06,0.8875229358673096,1.7389269260092988e-06,1.0504670143127441,0.9998529553413391,0.681113064289093,0.38402700424194336,0.00040480130701325834,2026-04-11T21:53:58Z
1201
+ 0.0,0.0,0.0022321429569274187,0.0022321429569274187,0.0022321429569274187,0.0,58.0,58.0,56.25,56.25,56.0,56.0,0.014292308245785534,0.045451034908866234,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1177,3.3288590908050537,6.436363636363637e-06,-0.0054,train,2537728.0,0.9876357316970825,1.0,0.0,1.0,0.0,0.9876357316970825,0.00018677377374842763,0.00018675869796425104,0.9876357316970825,0.00018677377374842763,0.9876357316970825,1.0,0.0,1.0,0.0,0.9876357316970825,0.00018677377374842763,0.9876357316970825,0.00018677377374842763,2.0,0.9979600310325623,0.0677042007446289,2.6926071643829346,0.014386294409632683,2026-04-11T21:54:03Z
1202
+ 0.01535926922224462,0.01535926922224462,0.0049019609577953815,0.0049019609577953815,0.020261230180040002,0.0,51.0,51.0,48.5,48.5,47.0,47.0,0.0684116561897099,0.04548965091133766,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1178,13.329391479492188,6.433333333333333e-06,0.0171,train,2539308.0,0.9343859553337097,1.0,0.0,1.0,0.0,0.9343859553337097,0.11850643157958984,0.11850643903017044,0.9343859553337097,0.11850643157958984,0.9343859553337097,1.0,0.0,1.0,0.0,0.9343859553337097,0.11850643157958984,0.9343859553337097,0.11850643157958984,2.0,0.9949454069137573,0.19431141018867493,1.6382932662963867,0.027483860030770302,2026-04-11T21:54:08Z
1203
+ 0.0019841270986944437,0.0019841270986944437,0.008163669612258673,0.008163669612258673,0.010147796710953116,0.0,63.0,63.0,61.625,61.625,61.0,61.0,0.03527964395470917,0.04552826691380908,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1179,4.12630033493042,6.430303030303031e-06,-0.0044,train,2541097.0,0.9966377019882202,1.0,0.0,1.0,0.0,0.9966377019882202,0.0008516657399013638,0.0008516703965142369,0.9966377019882202,0.0008516657399013638,0.9966377019882202,1.0,0.0,1.0,0.0,0.9966377019882202,0.0008516657399013638,0.9966377019882202,0.0008516657399013638,1.394187092781067,0.995635449886322,0.35570234060287476,1.0336610078811646,0.012902844697237015,2026-04-11T21:54:14Z
1204
+ 0.0007668711477890611,0.0007668711477890611,0.0015432098880410194,0.0015432098880410194,0.0023100810358300805,0.0,163.0,163.0,162.125,162.125,162.0,162.0,0.005278411292238161,0.045566882916280506,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1180,0.15215128660202026,6.427272727272728e-06,-0.0,train,2544066.0,0.9969795942306519,1.0,0.0,1.0,0.0,0.9969795942306519,5.755153688369319e-05,5.756055543315597e-05,0.9969795942306519,5.755153688369319e-05,0.9969795942306519,1.0,0.0,1.0,0.0,0.9969795942306519,5.755153688369319e-05,0.9969795942306519,5.755153688369319e-05,1.2257132530212402,0.9987585544586182,0.28496605157852173,1.255385160446167,0.00256105768494308,2026-04-11T21:54:20Z
1205
+ 0.0019841270986944437,0.0019841270986944437,0.002016128972172737,0.002016128972172737,0.004000256070867181,0.0,63.0,63.0,62.5,62.5,62.0,62.0,0.02104826516006142,0.04560549891875193,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1181,1.7918148040771484,6.424242424242425e-06,-0.0015,train,2545862.0,0.998000979423523,1.0,0.0,1.0,0.0,0.998000979423523,5.412446989794262e-05,5.414646147983149e-05,0.998000979423523,5.412446989794262e-05,0.998000979423523,1.0,0.0,1.0,0.0,0.998000979423523,5.412446989794262e-05,0.998000979423523,5.412446989794262e-05,1.2694493532180786,0.9972570538520813,0.4500918388366699,0.7983036041259766,0.008854641579091549,2026-04-11T21:54:25Z
1206
+ 0.0,0.0,0.0,0.0,0.0,0.0,210.0,210.0,210.0,210.0,210.0,210.0,0.0029908385331509635,0.045644114921223354,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1182,0.0,6.4212121212121215e-06,0.0,train,2548958.0,0.8545370101928711,1.0,0.0,0.8571428656578064,0.0,0.9969598650932312,0.0,0.0,0.8545370101928711,0.0,0.8545370101928711,1.0,0.0,0.8571428656578064,0.0,0.9969598650932312,0.0,0.8545370101928711,0.0,1.0694695711135864,0.9999819397926331,0.8643040657043457,0.14583073556423187,0.0003782480489462614,2026-04-11T21:54:31Z
1207
+ 0.0,0.0,0.0,0.0,0.0,0.0,73.0,73.0,73.0,73.0,73.0,73.0,0.007733863138128072,0.04568273092369478,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1183,0.0,6.418181818181819e-06,0.0,train,2550854.0,0.9984769821166992,1.0,0.0,1.0,0.0,0.9984769821166992,0.0,0.0,0.9984769821166992,0.0,0.9984769821166992,1.0,0.0,1.0,0.0,0.9984769821166992,0.0,0.9984769821166992,0.0,1.0240943431854248,1.0005358457565308,0.9428033828735352,0.0588974803686142,0.0008001961396075785,2026-04-11T21:54:36Z
1208
+ 0.011892712675035,0.011892712675035,0.03850760939531028,0.03850760939531028,0.05040032207034528,0.0,128.0,128.0,118.0,118.0,107.0,107.0,0.15224116947501898,0.0457213469261662,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1184,5.115583896636963,6.415151515151515e-06,0.0258,train,2553118.0,0.011638942174613476,1.0,0.0,1.0,0.0,0.011638942174613476,0.01774718053638935,0.01774718053638935,0.011638942174613476,0.01774718053638935,0.011638942174613476,1.0,0.0,1.0,0.0,0.011638942174613476,0.01774718053638935,0.011638942174613476,0.01774718053638935,2.0,0.9967266321182251,0.008854473941028118,4.726832389831543,0.0742255225777626,2026-04-11T21:54:42Z
1209
+ 0.0,0.0,0.004032257944345474,0.004032257944345474,0.004032257944345474,0.0,31.0,31.0,31.0,31.0,31.0,31.0,0.0018207905377494171,0.045759962928637626,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1185,0.43357840180397034,6.412121212121213e-06,-0.0021,train,2554678.0,0.9985817670822144,1.0,0.0,1.0,0.0,0.9985817670822144,2.903921813413035e-05,2.9045202609268017e-05,0.9985817670822144,2.903921813413035e-05,0.9985817670822144,1.0,0.0,1.0,0.0,0.9985817670822144,2.903921813413035e-05,0.9985817670822144,2.903921813413035e-05,1.0249570608139038,0.9974938631057739,0.34320297837257385,1.0694332122802734,0.004531952552497387,2026-04-11T21:54:46Z
1210
+ 0.0,0.0,0.0012886597542092204,0.0012886597542092204,0.0012886597542092204,0.0,99.0,99.0,98.0,98.0,97.0,97.0,0.011278128949925303,0.04579857893110905,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1186,1.6016870737075806,6.40909090909091e-06,-0.0023,train,2556862.0,0.9969850778579712,1.0,0.0,1.0,0.0,0.9969850778579712,9.93004723568447e-05,9.929558291332796e-05,0.9969850778579712,9.93004723568447e-05,0.9969850778579712,1.0,0.0,1.0,0.0,0.9969850778579712,9.93004723568447e-05,0.9969850778579712,9.93004723568447e-05,1.0727391242980957,0.9980478882789612,0.4588063955307007,0.7791270017623901,0.0031935099977999926,2026-04-11T21:54:52Z
1211
+ 0.0011467889416962862,0.0011467889416962862,0.0,0.0,0.0011467889416962862,0.0,109.0,109.0,109.0,109.0,109.0,109.0,0.010229390056338161,0.045837194933580475,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1187,0.6999150514602661,6.406060606060607e-06,0.0003,train,2559022.0,0.9984546899795532,1.0,0.0,1.0,0.0,0.9984546899795532,4.7120189265115187e-05,4.7120178351178765e-05,0.9984546899795532,4.7120189265115187e-05,0.9984546899795532,1.0,0.0,1.0,0.0,0.9984546899795532,4.7120189265115187e-05,0.9984546899795532,4.7120189265115187e-05,1.168613076210022,1.000537633895874,0.8960135579109192,0.15581762790679932,0.0013951655710116029,2026-04-11T21:54:57Z
1212
+ 0.006677350495010614,0.006677350495010614,0.0013736264081671834,0.0013736264081671834,0.008050976903177798,0.0,91.0,91.0,75.125,75.125,72.0,72.0,0.02158025815151632,0.0458758109360519,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1188,3.00974702835083,6.403030303030303e-06,0.0782,train,2560807.0,0.8669778108596802,1.0,0.0,1.0,0.0,0.8669778108596802,0.342582643032074,0.342582643032074,0.8669778108596802,0.342582643032074,0.8669778108596802,1.0,0.0,1.0,0.0,0.8669778108596802,0.342582643032074,0.8669778108596802,0.342582643032074,2.0,1.000046730041504,0.04143959656357765,3.183518409729004,0.014270931482315063,2026-04-11T21:55:02Z
1213
+ 0.001953125,0.001953125,0.022366520133800805,0.022366520133800805,0.024319645133800805,0.0,68.0,68.0,65.5,65.5,63.0,63.0,0.10702869668602943,0.04591442693852332,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1189,7.2858405113220215,6.4000000000000006e-06,0.0287,train,2562683.0,0.4144327640533447,1.0,0.0,1.0,0.0,0.4144327640533447,0.30921247601509094,0.30921247601509094,0.4144327640533447,0.30921247601509094,0.4144327640533447,1.0,0.0,1.0,0.0,0.4144327640533447,0.30921247601509094,0.4144327640533447,0.30921247601509094,2.0,1.0029542446136475,0.06300227344036102,2.764584541320801,0.035145364701747894,2026-04-11T21:55:07Z
1214
+ 0.0,0.0,0.0,0.0,0.0,0.0,31.0,31.0,31.0,31.0,31.0,31.0,0.0008445165294688195,0.04595304294099475,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1190,0.0,6.396969696969697e-06,0.0,train,2564211.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.0022221803665161,1.0001014471054077,0.9999852180480957,0.0022197323851287365,0.00010159891098737717,2026-04-11T21:55:11Z
1215
+ 0.0,0.0,0.0,0.0,0.0,0.0,91.0,91.0,91.0,91.0,91.0,91.0,0.00044536186032928526,0.04599165894346617,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1191,0.0,6.393939393939394e-06,0.0,train,2566107.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.0,0.9985920786857605,0.0,0.9985920786857605,1.0,0.0,1.0,0.0,0.9985920786857605,0.0,0.9985920786857605,0.0,1.001988172531128,1.0000598430633545,0.999610185623169,0.001986202783882618,6.144684448372573e-05,2026-04-11T21:55:16Z
1216
+ 0.021223400719463825,0.021223400719463825,0.005769230774603784,0.005769230774603784,0.02699263149406761,0.0,66.0,66.0,64.25,64.25,62.0,62.0,0.09414011146873236,0.046030274945937595,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1192,4.474715232849121,6.390909090909091e-06,0.0119,train,2568053.0,0.687461256980896,1.0,0.0,1.0,0.0,0.687461256980896,0.23647956550121307,0.23647956550121307,0.687461256980896,0.23647956550121307,0.687461256980896,1.0,0.0,1.0,0.0,0.687461256980896,0.23647956550121307,0.687461256980896,0.23647956550121307,1.4843707084655762,0.998434841632843,0.4403146207332611,0.8202657699584961,0.021633058786392212,2026-04-11T21:55:21Z
1217
+ 0.0,0.0,0.0,0.0,0.0,0.0,48.0,48.0,48.0,48.0,48.0,48.0,0.009338034316897392,0.04606889094840902,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1193,0.0,6.387878787878789e-06,0.0,train,2569709.0,0.9887858629226685,1.0,0.0,1.0,0.0,0.9887858629226685,0.0,0.0,0.9887858629226685,0.0,0.9887858629226685,1.0,0.0,1.0,0.0,0.9887858629226685,0.0,0.9887858629226685,0.0,1.0443377494812012,1.000608205795288,0.9549189805984497,0.04612874984741211,0.0012214966118335724,2026-04-11T21:55:26Z
1218
+ 0.0,0.0,0.0,0.0,0.0,0.0,182.0,182.0,181.125,181.125,181.0,181.0,0.006256086868233979,0.04610750695088044,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1194,0.1992160677909851,6.384848484848485e-06,-0.0001,train,2572614.0,0.8320591449737549,1.0,0.0,0.8333333134651184,0.0,0.9984710216522217,2.833232247212436e-05,2.3623691959073767e-05,0.8320591449737549,2.3610265998286195e-05,0.8320591449737549,1.0,0.0,0.8333333134651184,0.0,0.9984710216522217,2.833232247212436e-05,0.8320591449737549,2.3610265998286195e-05,1.209038496017456,0.9998201131820679,0.7053173184394836,0.3491075038909912,0.0009484349866397679,2026-04-11T21:55:32Z
1219
+ 0.004880740132648498,0.004880740132648498,0.007669382495805621,0.007669382495805621,0.012550122628454119,0.0,154.0,154.0,151.625,151.625,142.0,142.0,0.02665393566712737,0.04614612295335187,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1195,2.0341482162475586,6.381818181818182e-06,-0.0152,train,2575435.0,0.7980676889419556,1.0,0.0,0.800000011920929,0.0,0.9975845813751221,0.0008180320146493614,0.0006544221541844308,0.7980676889419556,0.0006544252391904593,0.7980676889419556,1.0,0.0,0.800000011920929,0.0,0.9975845813751221,0.0008180320146493614,0.7980676889419556,0.0006544252391904593,1.647913932800293,1.002147912979126,0.40257805585861206,0.9098663330078125,0.010841303505003452,2026-04-11T21:55:38Z
1220
+ 0.0035601977724581957,0.0035601977724581957,0.014889341779053211,0.014889341779053211,0.018449539551511407,0.0,106.0,106.0,103.625,103.625,100.0,100.0,0.03731701336801052,0.04618473895582329,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1196,9.221901893615723,6.37878787878788e-06,-0.01,train,2577632.0,0.9974844455718994,1.0,0.0,1.0,0.0,0.9974844455718994,0.0006991037516854703,0.0006991035188548267,0.9974844455718994,0.0006991037516854703,0.9974844455718994,1.0,0.0,1.0,0.0,0.9974844455718994,0.0006991037516854703,0.9974844455718994,0.0006991037516854703,2.0,0.9986835718154907,0.10909249633550644,2.215559244155884,0.015732292085886,2026-04-11T21:55:44Z
1221
+ 0.018585751531645656,0.018585751531645656,0.002659574383869767,0.002659574383869767,0.021245325915515423,0.0,47.0,47.0,35.75,35.75,31.0,31.0,0.05178397847339511,0.046223354958294716,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1197,24.279144287109375,6.375757575757576e-06,0.1338,train,2579110.0,0.9846537113189697,1.0,0.0,1.0,0.0,0.9846537113189697,0.032167691737413406,0.03216767683625221,0.9846537113189697,0.032167691737413406,0.9846537113189697,1.0,0.0,1.0,0.0,0.9846537113189697,0.032167691737413406,0.9846537113189697,0.032167691737413406,1.4753588438034058,0.995938777923584,0.04647897556424141,3.0687551498413086,0.037709400057792664,2026-04-11T21:55:48Z
1222
+ 0.001623376621864736,0.001623376621864736,0.025734331109561026,0.025734331109561026,0.027357707731425762,0.125,512.0,87.0,128.125,73.28572082519531,23.0,23.0,0.17751457169651985,0.04626197096076614,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1198,4.205373287200928,6.372727272727274e-06,-0.0433,train,2580935.0,0.042788296937942505,1.0,0.0,0.75,0.4629100561141968,0.042788296937942505,0.09419418126344681,0.09419417381286621,0.042788296937942505,0.09419418126344681,0.042788296937942505,1.0,0.0,0.75,0.4629100561141968,0.042788296937942505,0.09419418126344681,0.042788296937942505,0.09419418126344681,2.0,0.9972495436668396,0.022558048367500305,3.791663408279419,0.06149071082472801,2026-04-11T21:55:58Z
1223
+ 0.0,0.0,0.0,0.0,0.0,0.0,109.0,109.0,109.0,109.0,109.0,109.0,0.006159535580081865,0.046300586963237564,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1199,0.07112119346857071,6.3696969696969706e-06,0.0003,train,2583063.0,0.9984593391418457,1.0,0.0,1.0,0.0,0.9984593391418457,2.086405856971396e-06,2.0953264083800605e-06,0.9984593391418457,2.086405856971396e-06,0.9984593391418457,1.0,0.0,1.0,0.0,0.9984593391418457,2.086405856971396e-06,0.9984593391418457,2.086405856971396e-06,1.0703777074813843,1.0001569986343384,0.8332375884056091,0.18243646621704102,0.0007174843340180814,2026-04-11T21:56:04Z
1224
+ 0.0,0.0,0.0,0.0,0.0,0.0,24.0,24.0,24.0,24.0,24.0,24.0,0.006478953931946307,0.04633920296570899,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,1.0,1200,0.0,6.366666666666668e-06,0.0,train,2584463.0,0.9885647892951965,1.0,0.0,1.0,0.0,0.9885647892951965,0.0,0.0,0.9885647892951965,0.0,0.9885647892951965,1.0,0.0,1.0,0.0,0.9885647892951965,0.0,0.9885647892951965,0.0,1.0128138065338135,1.000182867050171,0.981205403804779,0.01897343248128891,0.0006689532892778516,2026-04-11T21:56:08Z
1225
+ ,,,,,,,,,,,,,0.04633920296570899,0.0,0.0,0.0,0.0,0.0,0.16346153846153846,487.9230769230769,423.6923076923077,250.20192307692307,198.65751765324518,58.92307692307692,58.92307692307692,0.01765880210754963,0.0,nan,2584463.0,0.4919216437981679,1.0,0.0,0.8193528331243075,0.28436795794046843,0.5284909640367215,0.49147624923632693,nan,0.4919216437981679,0.4640866976517897,0.4919216437981679,1.0,0.0,0.8193528331243075,0.28436795794046843,0.5284909640367215,0.49147624923632693,0.4919216437981679,0.4640866976517897,90.6551,1.147,1.286704604442303,1.0002888853733356,0.6044848309113429,0.5404733006770794,0.002731083811690601,0.143,,1200,,,,eval,,,,,,,,,,,,,,,,,,,,,,,,,,2026-04-11T21:57:39Z
1226
+ 0.024252045433968306,0.024252045433968306,0.007075471803545952,0.007075471803545952,0.03132751723751426,0.0,55.0,55.0,52.625,52.625,50.0,50.0,0.09002233669161797,0.04637781896818041,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1201,9.135161399841309,6.363636363636364e-06,0.0132,train,2586236.0,0.6550425291061401,1.0,0.0,1.0,0.0,0.6550425291061401,0.43107059597969055,0.43107059597969055,0.6550425291061401,0.43107059597969055,0.6550425291061401,1.0,0.0,1.0,0.0,0.6550425291061401,0.43107059597969055,0.6550425291061401,0.43107059597969055,1.7149361371994019,0.9945924282073975,0.030232258141040802,3.4988458156585693,0.03184327110648155,2026-04-11T21:57:46Z
metrics.jsonl CHANGED
@@ -1172,3 +1172,54 @@
1172
  {"timestamp_utc": "2026-04-11T21:50:08Z", "mode": "train", "global_step": 1150, "epoch": 0.04440840284213778, "loss": 0.0522, "grad_norm": 5.2525129318237305, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2483289.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8777114748954773, "rewards/meter/std": 0.16049861907958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777114748954773, "rewards/total_composite/std": 0.16049861907958984, "reward": 0.8777114748954773, "reward_std": 0.16049860417842865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024218173697590828, "sampling/sampling_logp_difference/max": 5.00484561920166, "sampling/importance_sampling_ratio/min": 0.006705376319587231, "sampling/importance_sampling_ratio/mean": 0.9977035522460938, "sampling/importance_sampling_ratio/max": 1.5273561477661133, "entropy": 0.05149026960134506, "clip_ratio/low_mean": 0.004067460540682077, "clip_ratio/low_min": 0.004067460540682077, "clip_ratio/high_mean": 0.009015594609081745, "clip_ratio/high_max": 0.009015594609081745, "clip_ratio/region_mean": 0.013083055149763823, "reward_total_mean": 0.8777114748954773, "reward_meter_mean": 0.8777114748954773, "reward_meter_std": 0.16049861907958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777114748954773, "reward_total_composite_std": 0.16049861907958984}
1173
  {"timestamp_utc": "2026-04-11T21:51:32Z", "mode": "eval", "global_step": 1150, "epoch": 0.04440840284213778, "eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154, "eval_num_tokens": 2483289.0, "eval_completions/mean_length": 216.47115384615384, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 447.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 197.60806157038763, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 396.38461538461536, "eval_rewards/meter/mean": 0.5331913347427661, "eval_rewards/meter/std": 0.45816060442190903, "eval_rewards/count_adherence/mean": 0.9469390053015488, "eval_rewards/count_adherence/std": 0.06932907207654072, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.511953374514213, "eval_rewards/total_composite/std": 0.440208015533594, "eval_reward": 0.511953374514213, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0024267814587801695, "eval_sampling/sampling_logp_difference/max": 0.5305701494216919, "eval_sampling/importance_sampling_ratio/min": 0.629830559858909, "eval_sampling/importance_sampling_ratio/mean": 1.0007386207580566, "eval_sampling/importance_sampling_ratio/max": 1.403554081916809, "eval_entropy": 0.017647026679836787, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.511953374514213, "eval_reward_meter_mean": 0.5331913347427661, "eval_reward_meter_std": 0.45816060442190903, "eval_reward_count_adherence_mean": 0.9469390053015488, "eval_reward_count_adherence_std": 0.06932907207654072, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.511953374514213, "eval_reward_total_composite_std": 0.440208015533594}
1174
  {"timestamp_utc": "2026-04-11T21:51:41Z", "mode": "train", "global_step": 1151, "epoch": 0.04444701884460921, "loss": 0.0504, "grad_norm": 3.7881219387054443, "learning_rate": 6.515151515151516e-06, "num_tokens": 2485639.0, "completions/mean_length": 122.75, "completions/min_length": 120.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.888770341873169, "rewards/meter/std": 0.21289391815662384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.888770341873169, "rewards/total_composite/std": 0.21289391815662384, "reward": 0.888770341873169, "reward_std": 0.21289391815662384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020240498706698418, "sampling/sampling_logp_difference/max": 6.1473388671875, "sampling/importance_sampling_ratio/min": 0.0021391669288277626, "sampling/importance_sampling_ratio/mean": 0.9987115263938904, "sampling/importance_sampling_ratio/max": 1.9864379167556763, "entropy": 0.013336000498384237, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.010218254523351789, "reward_total_mean": 0.888770341873169, "reward_meter_mean": 0.888770341873169, "reward_meter_std": 0.21289391815662384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.888770341873169, "reward_total_composite_std": 0.21289391815662384}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1172
  {"timestamp_utc": "2026-04-11T21:50:08Z", "mode": "train", "global_step": 1150, "epoch": 0.04440840284213778, "loss": 0.0522, "grad_norm": 5.2525129318237305, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2483289.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8777114748954773, "rewards/meter/std": 0.16049861907958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777114748954773, "rewards/total_composite/std": 0.16049861907958984, "reward": 0.8777114748954773, "reward_std": 0.16049860417842865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024218173697590828, "sampling/sampling_logp_difference/max": 5.00484561920166, "sampling/importance_sampling_ratio/min": 0.006705376319587231, "sampling/importance_sampling_ratio/mean": 0.9977035522460938, "sampling/importance_sampling_ratio/max": 1.5273561477661133, "entropy": 0.05149026960134506, "clip_ratio/low_mean": 0.004067460540682077, "clip_ratio/low_min": 0.004067460540682077, "clip_ratio/high_mean": 0.009015594609081745, "clip_ratio/high_max": 0.009015594609081745, "clip_ratio/region_mean": 0.013083055149763823, "reward_total_mean": 0.8777114748954773, "reward_meter_mean": 0.8777114748954773, "reward_meter_std": 0.16049861907958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777114748954773, "reward_total_composite_std": 0.16049861907958984}
1173
  {"timestamp_utc": "2026-04-11T21:51:32Z", "mode": "eval", "global_step": 1150, "epoch": 0.04440840284213778, "eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154, "eval_num_tokens": 2483289.0, "eval_completions/mean_length": 216.47115384615384, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 447.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 197.60806157038763, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 396.38461538461536, "eval_rewards/meter/mean": 0.5331913347427661, "eval_rewards/meter/std": 0.45816060442190903, "eval_rewards/count_adherence/mean": 0.9469390053015488, "eval_rewards/count_adherence/std": 0.06932907207654072, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.511953374514213, "eval_rewards/total_composite/std": 0.440208015533594, "eval_reward": 0.511953374514213, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0024267814587801695, "eval_sampling/sampling_logp_difference/max": 0.5305701494216919, "eval_sampling/importance_sampling_ratio/min": 0.629830559858909, "eval_sampling/importance_sampling_ratio/mean": 1.0007386207580566, "eval_sampling/importance_sampling_ratio/max": 1.403554081916809, "eval_entropy": 0.017647026679836787, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.511953374514213, "eval_reward_meter_mean": 0.5331913347427661, "eval_reward_meter_std": 0.45816060442190903, "eval_reward_count_adherence_mean": 0.9469390053015488, "eval_reward_count_adherence_std": 0.06932907207654072, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.511953374514213, "eval_reward_total_composite_std": 0.440208015533594}
1174
  {"timestamp_utc": "2026-04-11T21:51:41Z", "mode": "train", "global_step": 1151, "epoch": 0.04444701884460921, "loss": 0.0504, "grad_norm": 3.7881219387054443, "learning_rate": 6.515151515151516e-06, "num_tokens": 2485639.0, "completions/mean_length": 122.75, "completions/min_length": 120.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.888770341873169, "rewards/meter/std": 0.21289391815662384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.888770341873169, "rewards/total_composite/std": 0.21289391815662384, "reward": 0.888770341873169, "reward_std": 0.21289391815662384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020240498706698418, "sampling/sampling_logp_difference/max": 6.1473388671875, "sampling/importance_sampling_ratio/min": 0.0021391669288277626, "sampling/importance_sampling_ratio/mean": 0.9987115263938904, "sampling/importance_sampling_ratio/max": 1.9864379167556763, "entropy": 0.013336000498384237, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.010218254523351789, "reward_total_mean": 0.888770341873169, "reward_meter_mean": 0.888770341873169, "reward_meter_std": 0.21289391815662384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.888770341873169, "reward_total_composite_std": 0.21289391815662384}
1175
+ {"timestamp_utc": "2026-04-11T21:51:45Z", "mode": "train", "global_step": 1152, "epoch": 0.04448563484708063, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.512121212121213e-06, "num_tokens": 2487423.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.262973617296666e-05, "sampling/sampling_logp_difference/max": 0.002871689386665821, "sampling/importance_sampling_ratio/min": 0.9996728897094727, "sampling/importance_sampling_ratio/mean": 1.0000879764556885, "sampling/importance_sampling_ratio/max": 1.002875804901123, "entropy": 0.0009597428434062749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1176
+ {"timestamp_utc": "2026-04-11T21:51:50Z", "mode": "train", "global_step": 1153, "epoch": 0.044524250849552055, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2489239.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015291250019799918, "sampling/sampling_logp_difference/max": 0.014015945605933666, "sampling/importance_sampling_ratio/min": 0.9860818386077881, "sampling/importance_sampling_ratio/mean": 1.0000462532043457, "sampling/importance_sampling_ratio/max": 1.0045279264450073, "entropy": 0.00251059714355506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0}
1177
+ {"timestamp_utc": "2026-04-11T21:51:55Z", "mode": "train", "global_step": 1154, "epoch": 0.04456286685202348, "loss": 0.096, "grad_norm": 5.494170665740967, "learning_rate": 6.506060606060607e-06, "num_tokens": 2491171.0, "completions/mean_length": 65.5, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.13582952320575714, "rewards/meter/std": 0.19803810119628906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.13582952320575714, "rewards/total_composite/std": 0.19803810119628906, "reward": 0.13582952320575714, "reward_std": 0.19803808629512787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041303541511297226, "sampling/sampling_logp_difference/max": 4.165966033935547, "sampling/importance_sampling_ratio/min": 0.015514720231294632, "sampling/importance_sampling_ratio/mean": 0.9954270124435425, "sampling/importance_sampling_ratio/max": 1.913017988204956, "entropy": 0.0860102130100131, "clip_ratio/low_mean": 0.013047155574895442, "clip_ratio/low_min": 0.013047155574895442, "clip_ratio/high_mean": 0.008954269345849752, "clip_ratio/high_max": 0.008954269345849752, "clip_ratio/region_mean": 0.022001424920745194, "reward_total_mean": 0.13582952320575714, "reward_meter_mean": 0.13582952320575714, "reward_meter_std": 0.19803810119628906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.13582952320575714, "reward_total_composite_std": 0.19803810119628906}
1178
+ {"timestamp_utc": "2026-04-11T21:52:02Z", "mode": "train", "global_step": 1155, "epoch": 0.0446014828544949, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.503030303030303e-06, "num_tokens": 2493723.0, "completions/mean_length": 130.0, "completions/min_length": 130.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9970986843109131, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9970986843109131, "rewards/total_composite/std": 0.0, "reward": 0.9970986843109131, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001729379582684487, "sampling/sampling_logp_difference/max": 0.117275670170784, "sampling/importance_sampling_ratio/min": 0.8893400430679321, "sampling/importance_sampling_ratio/mean": 0.9999405145645142, "sampling/importance_sampling_ratio/max": 1.0045697689056396, "entropy": 0.0005436375031422358, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970986843109131, "reward_meter_mean": 0.9970986843109131, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9970986843109131, "reward_total_composite_std": 0.0}
1179
+ {"timestamp_utc": "2026-04-11T21:52:12Z", "mode": "train", "global_step": 1156, "epoch": 0.04464009885696633, "loss": -0.0346, "grad_norm": 3.3614652156829834, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2497463.0, "completions/mean_length": 304.5, "completions/min_length": 256.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 274.8571472167969, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.028450578451156616, "rewards/meter/std": 0.04791045933961868, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.2020305097103119, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.025792036205530167, "rewards/total_composite/std": 0.048944856971502304, "reward": 0.025792036205530167, "reward_std": 0.048944856971502304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024456597864627838, "sampling/sampling_logp_difference/max": 4.192187309265137, "sampling/importance_sampling_ratio/min": 0.015113191679120064, "sampling/importance_sampling_ratio/mean": 1.001096487045288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04777472233399749, "clip_ratio/low_mean": 0.014450150192715228, "clip_ratio/low_min": 0.014450150192715228, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.016403275192715228, "reward_total_mean": 0.025792036205530167, "reward_meter_mean": 0.028450578451156616, "reward_meter_std": 0.04791045933961868, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.2020305097103119, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.025792036205530167, "reward_total_composite_std": 0.048944856971502304}
1180
+ {"timestamp_utc": "2026-04-11T21:52:16Z", "mode": "train", "global_step": 1157, "epoch": 0.04467871485943775, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.496969696969697e-06, "num_tokens": 2499215.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0148831931874156, "sampling/sampling_logp_difference/max": 3.352174758911133, "sampling/importance_sampling_ratio/min": 0.03500813990831375, "sampling/importance_sampling_ratio/mean": 0.9938147664070129, "sampling/importance_sampling_ratio/max": 1.0112634897232056, "entropy": 0.006829831196228042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0}
1181
+ {"timestamp_utc": "2026-04-11T21:52:21Z", "mode": "train", "global_step": 1158, "epoch": 0.044717330861909176, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.493939393939395e-06, "num_tokens": 2501127.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9888595938682556, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888595938682556, "rewards/total_composite/std": 0.0, "reward": 0.9888595938682556, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00017435323388781399, "sampling/sampling_logp_difference/max": 0.0043810089118778706, "sampling/importance_sampling_ratio/min": 0.9968689680099487, "sampling/importance_sampling_ratio/mean": 1.000144600868225, "sampling/importance_sampling_ratio/max": 1.0043905973434448, "entropy": 0.0018330513557884842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888595938682556, "reward_meter_mean": 0.9888595938682556, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888595938682556, "reward_total_composite_std": 0.0}
1182
+ {"timestamp_utc": "2026-04-11T21:52:26Z", "mode": "train", "global_step": 1159, "epoch": 0.0447559468643806, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.490909090909091e-06, "num_tokens": 2503135.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013290751667227596, "sampling/sampling_logp_difference/max": 0.006528853438794613, "sampling/importance_sampling_ratio/min": 0.9994356036186218, "sampling/importance_sampling_ratio/mean": 1.0001311302185059, "sampling/importance_sampling_ratio/max": 1.0065501928329468, "entropy": 0.0013790328812319785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1183
+ {"timestamp_utc": "2026-04-11T21:52:31Z", "mode": "train", "global_step": 1160, "epoch": 0.044794562866852024, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.487878787878789e-06, "num_tokens": 2504759.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001226484455401078, "sampling/sampling_logp_difference/max": 0.005975149571895599, "sampling/importance_sampling_ratio/min": 0.9942424893379211, "sampling/importance_sampling_ratio/mean": 1.0000890493392944, "sampling/importance_sampling_ratio/max": 1.005993127822876, "entropy": 0.001158960752945859, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0}
1184
+ {"timestamp_utc": "2026-04-11T21:52:36Z", "mode": "train", "global_step": 1161, "epoch": 0.04483317886932345, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.484848484848485e-06, "num_tokens": 2506847.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00014731692499481142, "sampling/sampling_logp_difference/max": 0.005519423168152571, "sampling/importance_sampling_ratio/min": 0.9944958090782166, "sampling/importance_sampling_ratio/mean": 1.0000895261764526, "sampling/importance_sampling_ratio/max": 1.0041718482971191, "entropy": 0.0013596047574537806, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0}
1185
+ {"timestamp_utc": "2026-04-11T21:52:41Z", "mode": "train", "global_step": 1162, "epoch": 0.04487179487179487, "loss": -0.0067, "grad_norm": 9.667710304260254, "learning_rate": 6.481818181818182e-06, "num_tokens": 2508562.0, "completions/mean_length": 54.375, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9774544835090637, "rewards/meter/std": 0.007204634603112936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9774544835090637, "rewards/total_composite/std": 0.007204634603112936, "reward": 0.9774544835090637, "reward_std": 0.007204628549516201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024649647995829582, "sampling/sampling_logp_difference/max": 3.8922629356384277, "sampling/importance_sampling_ratio/min": 0.02039913274347782, "sampling/importance_sampling_ratio/mean": 0.9978480339050293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028170868754386902, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.004590633558109403, "clip_ratio/high_max": 0.004590633558109403, "clip_ratio/region_mean": 0.009220263222232461, "reward_total_mean": 0.9774544835090637, "reward_meter_mean": 0.9774544835090637, "reward_meter_std": 0.007204634603112936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9774544835090637, "reward_total_composite_std": 0.007204634603112936}
1186
+ {"timestamp_utc": "2026-04-11T21:52:47Z", "mode": "train", "global_step": 1163, "epoch": 0.044910410874266296, "loss": -0.0003, "grad_norm": 0.0491514727473259, "learning_rate": 6.478787878787879e-06, "num_tokens": 2511042.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9984663724899292, "rewards/meter/std": 3.918135462299688e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984663724899292, "rewards/total_composite/std": 3.918135462299688e-06, "reward": 0.9984663724899292, "reward_std": 3.918005404557334e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012445234460756183, "sampling/sampling_logp_difference/max": 0.6710500717163086, "sampling/importance_sampling_ratio/min": 0.8294064402580261, "sampling/importance_sampling_ratio/mean": 1.000813603401184, "sampling/importance_sampling_ratio/max": 1.9562904834747314, "entropy": 0.006222540338058025, "clip_ratio/low_mean": 0.0008620689623057842, "clip_ratio/low_min": 0.0008620689623057842, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0008620689623057842, "reward_total_mean": 0.9984663724899292, "reward_meter_mean": 0.9984663724899292, "reward_meter_std": 3.918135462299688e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984663724899292, "reward_total_composite_std": 3.918135462299688e-06}
1187
+ {"timestamp_utc": "2026-04-11T21:52:51Z", "mode": "train", "global_step": 1164, "epoch": 0.04494902687673772, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.475757575757576e-06, "num_tokens": 2512906.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.520409755874425e-05, "sampling/sampling_logp_difference/max": 0.0016170135932043195, "sampling/importance_sampling_ratio/min": 0.9988471865653992, "sampling/importance_sampling_ratio/mean": 1.000075340270996, "sampling/importance_sampling_ratio/max": 1.0016183853149414, "entropy": 0.0006748539672116749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1188
+ {"timestamp_utc": "2026-04-11T21:52:56Z", "mode": "train", "global_step": 1165, "epoch": 0.044987642879209144, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.472727272727272e-06, "num_tokens": 2514546.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018705571710597724, "sampling/sampling_logp_difference/max": 0.003838915377855301, "sampling/importance_sampling_ratio/min": 0.996885359287262, "sampling/importance_sampling_ratio/mean": 1.0001657009124756, "sampling/importance_sampling_ratio/max": 1.003846287727356, "entropy": 0.0014549151237588376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0}
1189
+ {"timestamp_utc": "2026-04-11T21:53:01Z", "mode": "train", "global_step": 1166, "epoch": 0.04502625888168057, "loss": 0.0454, "grad_norm": 190.9034881591797, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2516434.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.996626615524292, "rewards/meter/std": 0.001510909991338849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.996626615524292, "rewards/total_composite/std": 0.001510909991338849, "reward": 0.996626615524292, "reward_std": 0.0015109025407582521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025698505342006683, "sampling/sampling_logp_difference/max": 1.1933188438415527, "sampling/importance_sampling_ratio/min": 0.3032132685184479, "sampling/importance_sampling_ratio/mean": 0.9995715618133545, "sampling/importance_sampling_ratio/max": 1.7860766649246216, "entropy": 0.059585667215287685, "clip_ratio/low_mean": 0.006657268386334181, "clip_ratio/low_min": 0.006657268386334181, "clip_ratio/high_mean": 0.008513708598911762, "clip_ratio/high_max": 0.008513708598911762, "clip_ratio/region_mean": 0.015170976985245943, "reward_total_mean": 0.996626615524292, "reward_meter_mean": 0.996626615524292, "reward_meter_std": 0.001510909991338849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.996626615524292, "reward_total_composite_std": 0.001510909991338849}
1190
+ {"timestamp_utc": "2026-04-11T21:53:06Z", "mode": "train", "global_step": 1167, "epoch": 0.04506487488415199, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.466666666666667e-06, "num_tokens": 2518002.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013065863167867064, "sampling/sampling_logp_difference/max": 0.18986795842647552, "sampling/importance_sampling_ratio/min": 0.8270683288574219, "sampling/importance_sampling_ratio/mean": 0.9993685483932495, "sampling/importance_sampling_ratio/max": 1.032031774520874, "entropy": 0.006112938834121451, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0}
1191
+ {"timestamp_utc": "2026-04-11T21:53:11Z", "mode": "train", "global_step": 1168, "epoch": 0.04510349088662342, "loss": 0.0287, "grad_norm": 7.1253252029418945, "learning_rate": 6.463636363636364e-06, "num_tokens": 2520049.0, "completions/mean_length": 81.875, "completions/min_length": 80.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.03959798067808151, "rewards/meter/std": 0.011728010140359402, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.038292575627565384, "rewards/total_composite/std": 0.01326083205640316, "reward": 0.038292575627565384, "reward_std": 0.01326083205640316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028142500668764114, "sampling/sampling_logp_difference/max": 4.7991814613342285, "sampling/importance_sampling_ratio/min": 0.008236486464738846, "sampling/importance_sampling_ratio/mean": 0.9953240752220154, "sampling/importance_sampling_ratio/max": 1.6731361150741577, "entropy": 0.03918551583774388, "clip_ratio/low_mean": 0.014004629920236766, "clip_ratio/low_min": 0.014004629920236766, "clip_ratio/high_mean": 0.004611280397512019, "clip_ratio/high_max": 0.004611280397512019, "clip_ratio/region_mean": 0.018615910317748785, "reward_total_mean": 0.038292575627565384, "reward_meter_mean": 0.03959798067808151, "reward_meter_std": 0.011728010140359402, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.038292575627565384, "reward_total_composite_std": 0.01326083205640316}
1192
+ {"timestamp_utc": "2026-04-11T21:53:15Z", "mode": "train", "global_step": 1169, "epoch": 0.04514210688909484, "loss": -0.044, "grad_norm": 4.53950834274292, "learning_rate": 6.460606060606061e-06, "num_tokens": 2521727.0, "completions/mean_length": 58.75, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9976530075073242, "rewards/meter/std": 0.0003900358860846609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9976530075073242, "rewards/total_composite/std": 0.0003900358860846609, "reward": 0.9976530075073242, "reward_std": 0.00039003457641229033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013675752095878124, "sampling/sampling_logp_difference/max": 0.7795699834823608, "sampling/importance_sampling_ratio/min": 0.4586032032966614, "sampling/importance_sampling_ratio/mean": 0.9979230165481567, "sampling/importance_sampling_ratio/max": 1.4813544750213623, "entropy": 0.0419846111908555, "clip_ratio/low_mean": 0.006818181602284312, "clip_ratio/low_min": 0.006818181602284312, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.00883431057445705, "reward_total_mean": 0.9976530075073242, "reward_meter_mean": 0.9976530075073242, "reward_meter_std": 0.0003900358860846609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9976530075073242, "reward_total_composite_std": 0.0003900358860846609}
1193
+ {"timestamp_utc": "2026-04-11T21:53:26Z", "mode": "train", "global_step": 1170, "epoch": 0.045180722891566265, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.457575757575758e-06, "num_tokens": 2523215.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9969618916511536, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545387387275696, "rewards/total_composite/std": 0.0, "reward": 0.8545387387275696, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545387387275696, "reward_meter_mean": 0.9969618916511536, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545387387275696, "reward_total_composite_std": 0.0}
1194
+ {"timestamp_utc": "2026-04-11T21:53:30Z", "mode": "train", "global_step": 1171, "epoch": 0.04521933889403769, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.454545454545456e-06, "num_tokens": 2524991.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013418152229860425, "sampling/sampling_logp_difference/max": 0.004571585915982723, "sampling/importance_sampling_ratio/min": 0.999534547328949, "sampling/importance_sampling_ratio/mean": 1.0001263618469238, "sampling/importance_sampling_ratio/max": 1.0045820474624634, "entropy": 0.0013301519793458283, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1195
+ {"timestamp_utc": "2026-04-11T21:53:36Z", "mode": "train", "global_step": 1172, "epoch": 0.04525795489650911, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.451515151515152e-06, "num_tokens": 2527167.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002919211983680725, "sampling/sampling_logp_difference/max": 0.04608858376741409, "sampling/importance_sampling_ratio/min": 0.9549573659896851, "sampling/importance_sampling_ratio/mean": 1.00016450881958, "sampling/importance_sampling_ratio/max": 1.0190335512161255, "entropy": 0.00159282027016161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0}
1196
+ {"timestamp_utc": "2026-04-11T21:53:41Z", "mode": "train", "global_step": 1173, "epoch": 0.04529657089898054, "loss": 0.0337, "grad_norm": 20.588041305541992, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2528827.0, "completions/mean_length": 48.5, "completions/min_length": 48.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9464411735534668, "rewards/meter/std": 0.11976895481348038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9464411735534668, "rewards/total_composite/std": 0.11976895481348038, "reward": 0.9464411735534668, "reward_std": 0.11976895481348038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008157458156347275, "sampling/sampling_logp_difference/max": 1.4072527885437012, "sampling/importance_sampling_ratio/min": 0.244814932346344, "sampling/importance_sampling_ratio/mean": 1.0024529695510864, "sampling/importance_sampling_ratio/max": 1.9219340085983276, "entropy": 0.018341065326239914, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.0026041667442768812, "clip_ratio/high_max": 0.0026041667442768812, "clip_ratio/region_mean": 0.007411859231069684, "reward_total_mean": 0.9464411735534668, "reward_meter_mean": 0.9464411735534668, "reward_meter_std": 0.11976895481348038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9464411735534668, "reward_total_composite_std": 0.11976895481348038}
1197
+ {"timestamp_utc": "2026-04-11T21:53:45Z", "mode": "train", "global_step": 1174, "epoch": 0.04533518690145196, "loss": 0.032, "grad_norm": 12.414958953857422, "learning_rate": 6.445454545454546e-06, "num_tokens": 2530280.0, "completions/mean_length": 30.625, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9770581126213074, "rewards/meter/std": 0.03885127976536751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9770581126213074, "rewards/total_composite/std": 0.03885127976536751, "reward": 0.9770581126213074, "reward_std": 0.03885127976536751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00510073360055685, "sampling/sampling_logp_difference/max": 0.34801721572875977, "sampling/importance_sampling_ratio/min": 0.7204657793045044, "sampling/importance_sampling_ratio/mean": 1.0016231536865234, "sampling/importance_sampling_ratio/max": 1.4162566661834717, "entropy": 0.015783087466843426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/region_mean": 0.004166666883975267, "reward_total_mean": 0.9770581126213074, "reward_meter_mean": 0.9770581126213074, "reward_meter_std": 0.03885127976536751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9770581126213074, "reward_total_composite_std": 0.03885127976536751}
1198
+ {"timestamp_utc": "2026-04-11T21:53:50Z", "mode": "train", "global_step": 1175, "epoch": 0.045373802903923385, "loss": 0.0157, "grad_norm": 10.53234577178955, "learning_rate": 6.442424242424243e-06, "num_tokens": 2531910.0, "completions/mean_length": 62.75, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.98736572265625, "rewards/meter/std": 0.02987661585211754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.98736572265625, "rewards/total_composite/std": 0.02987661585211754, "reward": 0.98736572265625, "reward_std": 0.02987661026418209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009396832436323166, "sampling/sampling_logp_difference/max": 1.0190973281860352, "sampling/importance_sampling_ratio/min": 0.36092060804367065, "sampling/importance_sampling_ratio/mean": 1.0028626918792725, "sampling/importance_sampling_ratio/max": 1.5600427389144897, "entropy": 0.03662474290467799, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0059843831695616245, "clip_ratio/high_max": 0.0059843831695616245, "clip_ratio/region_mean": 0.007937508169561625, "reward_total_mean": 0.98736572265625, "reward_meter_mean": 0.98736572265625, "reward_meter_std": 0.02987661585211754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.98736572265625, "reward_total_composite_std": 0.02987661585211754}
1199
+ {"timestamp_utc": "2026-04-11T21:53:58Z", "mode": "train", "global_step": 1176, "epoch": 0.04541241890639481, "loss": 0.0004, "grad_norm": 0.008205018006265163, "learning_rate": 6.43939393939394e-06, "num_tokens": 2536102.0, "completions/mean_length": 307.0, "completions/min_length": 307.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 307.0, "completions/min_terminated_length": 307.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9984632730484009, "rewards/meter/std": 1.9595856883825036e-06, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8875229358673096, "rewards/total_composite/std": 1.7389269260092988e-06, "reward": 0.8875229358673096, "reward_std": 1.7291217773163226e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00040480130701325834, "sampling/sampling_logp_difference/max": 0.38402700424194336, "sampling/importance_sampling_ratio/min": 0.681113064289093, "sampling/importance_sampling_ratio/mean": 0.9998529553413391, "sampling/importance_sampling_ratio/max": 1.0504670143127441, "entropy": 0.001436470149201341, "clip_ratio/low_mean": 0.0004071661096531898, "clip_ratio/low_min": 0.0004071661096531898, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0004071661096531898, "reward_total_mean": 0.8875229358673096, "reward_meter_mean": 0.9984632730484009, "reward_meter_std": 1.9595856883825036e-06, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8875229358673096, "reward_total_composite_std": 1.7389269260092988e-06}
1200
+ {"timestamp_utc": "2026-04-11T21:54:03Z", "mode": "train", "global_step": 1177, "epoch": 0.045451034908866234, "loss": -0.0054, "grad_norm": 3.3288590908050537, "learning_rate": 6.436363636363637e-06, "num_tokens": 2537728.0, "completions/mean_length": 56.25, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9876357316970825, "rewards/meter/std": 0.00018677377374842763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9876357316970825, "rewards/total_composite/std": 0.00018677377374842763, "reward": 0.9876357316970825, "reward_std": 0.00018675869796425104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014386294409632683, "sampling/sampling_logp_difference/max": 2.6926071643829346, "sampling/importance_sampling_ratio/min": 0.0677042007446289, "sampling/importance_sampling_ratio/mean": 0.9979600310325623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014292308245785534, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0022321429569274187, "reward_total_mean": 0.9876357316970825, "reward_meter_mean": 0.9876357316970825, "reward_meter_std": 0.00018677377374842763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9876357316970825, "reward_total_composite_std": 0.00018677377374842763}
1201
+ {"timestamp_utc": "2026-04-11T21:54:08Z", "mode": "train", "global_step": 1178, "epoch": 0.04548965091133766, "loss": 0.0171, "grad_norm": 13.329391479492188, "learning_rate": 6.433333333333333e-06, "num_tokens": 2539308.0, "completions/mean_length": 48.5, "completions/min_length": 47.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9343859553337097, "rewards/meter/std": 0.11850643157958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9343859553337097, "rewards/total_composite/std": 0.11850643157958984, "reward": 0.9343859553337097, "reward_std": 0.11850643903017044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027483860030770302, "sampling/sampling_logp_difference/max": 1.6382932662963867, "sampling/importance_sampling_ratio/min": 0.19431141018867493, "sampling/importance_sampling_ratio/mean": 0.9949454069137573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0684116561897099, "clip_ratio/low_mean": 0.0049019609577953815, "clip_ratio/low_min": 0.0049019609577953815, "clip_ratio/high_mean": 0.01535926922224462, "clip_ratio/high_max": 0.01535926922224462, "clip_ratio/region_mean": 0.020261230180040002, "reward_total_mean": 0.9343859553337097, "reward_meter_mean": 0.9343859553337097, "reward_meter_std": 0.11850643157958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9343859553337097, "reward_total_composite_std": 0.11850643157958984}
1202
+ {"timestamp_utc": "2026-04-11T21:54:14Z", "mode": "train", "global_step": 1179, "epoch": 0.04552826691380908, "loss": -0.0044, "grad_norm": 4.12630033493042, "learning_rate": 6.430303030303031e-06, "num_tokens": 2541097.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9966377019882202, "rewards/meter/std": 0.0008516657399013638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966377019882202, "rewards/total_composite/std": 0.0008516657399013638, "reward": 0.9966377019882202, "reward_std": 0.0008516703965142369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012902844697237015, "sampling/sampling_logp_difference/max": 1.0336610078811646, "sampling/importance_sampling_ratio/min": 0.35570234060287476, "sampling/importance_sampling_ratio/mean": 0.995635449886322, "sampling/importance_sampling_ratio/max": 1.394187092781067, "entropy": 0.03527964395470917, "clip_ratio/low_mean": 0.008163669612258673, "clip_ratio/low_min": 0.008163669612258673, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.010147796710953116, "reward_total_mean": 0.9966377019882202, "reward_meter_mean": 0.9966377019882202, "reward_meter_std": 0.0008516657399013638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966377019882202, "reward_total_composite_std": 0.0008516657399013638}
1203
+ {"timestamp_utc": "2026-04-11T21:54:20Z", "mode": "train", "global_step": 1180, "epoch": 0.045566882916280506, "loss": -0.0, "grad_norm": 0.15215128660202026, "learning_rate": 6.427272727272728e-06, "num_tokens": 2544066.0, "completions/mean_length": 162.125, "completions/min_length": 162.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.125, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9969795942306519, "rewards/meter/std": 5.755153688369319e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969795942306519, "rewards/total_composite/std": 5.755153688369319e-05, "reward": 0.9969795942306519, "reward_std": 5.756055543315597e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00256105768494308, "sampling/sampling_logp_difference/max": 1.255385160446167, "sampling/importance_sampling_ratio/min": 0.28496605157852173, "sampling/importance_sampling_ratio/mean": 0.9987585544586182, "sampling/importance_sampling_ratio/max": 1.2257132530212402, "entropy": 0.005278411292238161, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.0007668711477890611, "clip_ratio/high_max": 0.0007668711477890611, "clip_ratio/region_mean": 0.0023100810358300805, "reward_total_mean": 0.9969795942306519, "reward_meter_mean": 0.9969795942306519, "reward_meter_std": 5.755153688369319e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969795942306519, "reward_total_composite_std": 5.755153688369319e-05}
1204
+ {"timestamp_utc": "2026-04-11T21:54:25Z", "mode": "train", "global_step": 1181, "epoch": 0.04560549891875193, "loss": -0.0015, "grad_norm": 1.7918148040771484, "learning_rate": 6.424242424242425e-06, "num_tokens": 2545862.0, "completions/mean_length": 62.5, "completions/min_length": 62.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.998000979423523, "rewards/meter/std": 5.412446989794262e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998000979423523, "rewards/total_composite/std": 5.412446989794262e-05, "reward": 0.998000979423523, "reward_std": 5.414646147983149e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008854641579091549, "sampling/sampling_logp_difference/max": 0.7983036041259766, "sampling/importance_sampling_ratio/min": 0.4500918388366699, "sampling/importance_sampling_ratio/mean": 0.9972570538520813, "sampling/importance_sampling_ratio/max": 1.2694493532180786, "entropy": 0.02104826516006142, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.004000256070867181, "reward_total_mean": 0.998000979423523, "reward_meter_mean": 0.998000979423523, "reward_meter_std": 5.412446989794262e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998000979423523, "reward_total_composite_std": 5.412446989794262e-05}
1205
+ {"timestamp_utc": "2026-04-11T21:54:31Z", "mode": "train", "global_step": 1182, "epoch": 0.045644114921223354, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2548958.0, "completions/mean_length": 210.0, "completions/min_length": 210.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 210.0, "completions/min_terminated_length": 210.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9969598650932312, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545370101928711, "rewards/total_composite/std": 0.0, "reward": 0.8545370101928711, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003782480489462614, "sampling/sampling_logp_difference/max": 0.14583073556423187, "sampling/importance_sampling_ratio/min": 0.8643040657043457, "sampling/importance_sampling_ratio/mean": 0.9999819397926331, "sampling/importance_sampling_ratio/max": 1.0694695711135864, "entropy": 0.0029908385331509635, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545370101928711, "reward_meter_mean": 0.9969598650932312, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545370101928711, "reward_total_composite_std": 0.0}
1206
+ {"timestamp_utc": "2026-04-11T21:54:36Z", "mode": "train", "global_step": 1183, "epoch": 0.04568273092369478, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.418181818181819e-06, "num_tokens": 2550854.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9984769821166992, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984769821166992, "rewards/total_composite/std": 0.0, "reward": 0.9984769821166992, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008001961396075785, "sampling/sampling_logp_difference/max": 0.0588974803686142, "sampling/importance_sampling_ratio/min": 0.9428033828735352, "sampling/importance_sampling_ratio/mean": 1.0005358457565308, "sampling/importance_sampling_ratio/max": 1.0240943431854248, "entropy": 0.007733863138128072, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984769821166992, "reward_meter_mean": 0.9984769821166992, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984769821166992, "reward_total_composite_std": 0.0}
1207
+ {"timestamp_utc": "2026-04-11T21:54:42Z", "mode": "train", "global_step": 1184, "epoch": 0.0457213469261662, "loss": 0.0258, "grad_norm": 5.115583896636963, "learning_rate": 6.415151515151515e-06, "num_tokens": 2553118.0, "completions/mean_length": 118.0, "completions/min_length": 107.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.011638942174613476, "rewards/meter/std": 0.01774718053638935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.011638942174613476, "rewards/total_composite/std": 0.01774718053638935, "reward": 0.011638942174613476, "reward_std": 0.01774718053638935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0742255225777626, "sampling/sampling_logp_difference/max": 4.726832389831543, "sampling/importance_sampling_ratio/min": 0.008854473941028118, "sampling/importance_sampling_ratio/mean": 0.9967266321182251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15224116947501898, "clip_ratio/low_mean": 0.03850760939531028, "clip_ratio/low_min": 0.03850760939531028, "clip_ratio/high_mean": 0.011892712675035, "clip_ratio/high_max": 0.011892712675035, "clip_ratio/region_mean": 0.05040032207034528, "reward_total_mean": 0.011638942174613476, "reward_meter_mean": 0.011638942174613476, "reward_meter_std": 0.01774718053638935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.011638942174613476, "reward_total_composite_std": 0.01774718053638935}
1208
+ {"timestamp_utc": "2026-04-11T21:54:46Z", "mode": "train", "global_step": 1185, "epoch": 0.045759962928637626, "loss": -0.0021, "grad_norm": 0.43357840180397034, "learning_rate": 6.412121212121213e-06, "num_tokens": 2554678.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985817670822144, "rewards/meter/std": 2.903921813413035e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985817670822144, "rewards/total_composite/std": 2.903921813413035e-05, "reward": 0.9985817670822144, "reward_std": 2.9045202609268017e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004531952552497387, "sampling/sampling_logp_difference/max": 1.0694332122802734, "sampling/importance_sampling_ratio/min": 0.34320297837257385, "sampling/importance_sampling_ratio/mean": 0.9974938631057739, "sampling/importance_sampling_ratio/max": 1.0249570608139038, "entropy": 0.0018207905377494171, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004032257944345474, "reward_total_mean": 0.9985817670822144, "reward_meter_mean": 0.9985817670822144, "reward_meter_std": 2.903921813413035e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985817670822144, "reward_total_composite_std": 2.903921813413035e-05}
1209
+ {"timestamp_utc": "2026-04-11T21:54:52Z", "mode": "train", "global_step": 1186, "epoch": 0.04579857893110905, "loss": -0.0023, "grad_norm": 1.6016870737075806, "learning_rate": 6.40909090909091e-06, "num_tokens": 2556862.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9969850778579712, "rewards/meter/std": 9.93004723568447e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969850778579712, "rewards/total_composite/std": 9.93004723568447e-05, "reward": 0.9969850778579712, "reward_std": 9.929558291332796e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0031935099977999926, "sampling/sampling_logp_difference/max": 0.7791270017623901, "sampling/importance_sampling_ratio/min": 0.4588063955307007, "sampling/importance_sampling_ratio/mean": 0.9980478882789612, "sampling/importance_sampling_ratio/max": 1.0727391242980957, "entropy": 0.011278128949925303, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012886597542092204, "reward_total_mean": 0.9969850778579712, "reward_meter_mean": 0.9969850778579712, "reward_meter_std": 9.93004723568447e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969850778579712, "reward_total_composite_std": 9.93004723568447e-05}
1210
+ {"timestamp_utc": "2026-04-11T21:54:57Z", "mode": "train", "global_step": 1187, "epoch": 0.045837194933580475, "loss": 0.0003, "grad_norm": 0.6999150514602661, "learning_rate": 6.406060606060607e-06, "num_tokens": 2559022.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984546899795532, "rewards/meter/std": 4.7120189265115187e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984546899795532, "rewards/total_composite/std": 4.7120189265115187e-05, "reward": 0.9984546899795532, "reward_std": 4.7120178351178765e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013951655710116029, "sampling/sampling_logp_difference/max": 0.15581762790679932, "sampling/importance_sampling_ratio/min": 0.8960135579109192, "sampling/importance_sampling_ratio/mean": 1.000537633895874, "sampling/importance_sampling_ratio/max": 1.168613076210022, "entropy": 0.010229390056338161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0011467889416962862, "clip_ratio/high_max": 0.0011467889416962862, "clip_ratio/region_mean": 0.0011467889416962862, "reward_total_mean": 0.9984546899795532, "reward_meter_mean": 0.9984546899795532, "reward_meter_std": 4.7120189265115187e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984546899795532, "reward_total_composite_std": 4.7120189265115187e-05}
1211
+ {"timestamp_utc": "2026-04-11T21:55:02Z", "mode": "train", "global_step": 1188, "epoch": 0.0458758109360519, "loss": 0.0782, "grad_norm": 3.00974702835083, "learning_rate": 6.403030303030303e-06, "num_tokens": 2560807.0, "completions/mean_length": 75.125, "completions/min_length": 72.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8669778108596802, "rewards/meter/std": 0.342582643032074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8669778108596802, "rewards/total_composite/std": 0.342582643032074, "reward": 0.8669778108596802, "reward_std": 0.342582643032074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014270931482315063, "sampling/sampling_logp_difference/max": 3.183518409729004, "sampling/importance_sampling_ratio/min": 0.04143959656357765, "sampling/importance_sampling_ratio/mean": 1.000046730041504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02158025815151632, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.006677350495010614, "clip_ratio/high_max": 0.006677350495010614, "clip_ratio/region_mean": 0.008050976903177798, "reward_total_mean": 0.8669778108596802, "reward_meter_mean": 0.8669778108596802, "reward_meter_std": 0.342582643032074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8669778108596802, "reward_total_composite_std": 0.342582643032074}
1212
+ {"timestamp_utc": "2026-04-11T21:55:07Z", "mode": "train", "global_step": 1189, "epoch": 0.04591442693852332, "loss": 0.0287, "grad_norm": 7.2858405113220215, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2562683.0, "completions/mean_length": 65.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.4144327640533447, "rewards/meter/std": 0.30921247601509094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4144327640533447, "rewards/total_composite/std": 0.30921247601509094, "reward": 0.4144327640533447, "reward_std": 0.30921247601509094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035145364701747894, "sampling/sampling_logp_difference/max": 2.764584541320801, "sampling/importance_sampling_ratio/min": 0.06300227344036102, "sampling/importance_sampling_ratio/mean": 1.0029542446136475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10702869668602943, "clip_ratio/low_mean": 0.022366520133800805, "clip_ratio/low_min": 0.022366520133800805, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.024319645133800805, "reward_total_mean": 0.4144327640533447, "reward_meter_mean": 0.4144327640533447, "reward_meter_std": 0.30921247601509094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4144327640533447, "reward_total_composite_std": 0.30921247601509094}
1213
+ {"timestamp_utc": "2026-04-11T21:55:11Z", "mode": "train", "global_step": 1190, "epoch": 0.04595304294099475, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.396969696969697e-06, "num_tokens": 2564211.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010159891098737717, "sampling/sampling_logp_difference/max": 0.0022197323851287365, "sampling/importance_sampling_ratio/min": 0.9999852180480957, "sampling/importance_sampling_ratio/mean": 1.0001014471054077, "sampling/importance_sampling_ratio/max": 1.0022221803665161, "entropy": 0.0008445165294688195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1214
+ {"timestamp_utc": "2026-04-11T21:55:16Z", "mode": "train", "global_step": 1191, "epoch": 0.04599165894346617, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.393939393939394e-06, "num_tokens": 2566107.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.144684448372573e-05, "sampling/sampling_logp_difference/max": 0.001986202783882618, "sampling/importance_sampling_ratio/min": 0.999610185623169, "sampling/importance_sampling_ratio/mean": 1.0000598430633545, "sampling/importance_sampling_ratio/max": 1.001988172531128, "entropy": 0.00044536186032928526, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0}
1215
+ {"timestamp_utc": "2026-04-11T21:55:21Z", "mode": "train", "global_step": 1192, "epoch": 0.046030274945937595, "loss": 0.0119, "grad_norm": 4.474715232849121, "learning_rate": 6.390909090909091e-06, "num_tokens": 2568053.0, "completions/mean_length": 64.25, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.687461256980896, "rewards/meter/std": 0.23647956550121307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.687461256980896, "rewards/total_composite/std": 0.23647956550121307, "reward": 0.687461256980896, "reward_std": 0.23647956550121307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021633058786392212, "sampling/sampling_logp_difference/max": 0.8202657699584961, "sampling/importance_sampling_ratio/min": 0.4403146207332611, "sampling/importance_sampling_ratio/mean": 0.998434841632843, "sampling/importance_sampling_ratio/max": 1.4843707084655762, "entropy": 0.09414011146873236, "clip_ratio/low_mean": 0.005769230774603784, "clip_ratio/low_min": 0.005769230774603784, "clip_ratio/high_mean": 0.021223400719463825, "clip_ratio/high_max": 0.021223400719463825, "clip_ratio/region_mean": 0.02699263149406761, "reward_total_mean": 0.687461256980896, "reward_meter_mean": 0.687461256980896, "reward_meter_std": 0.23647956550121307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.687461256980896, "reward_total_composite_std": 0.23647956550121307}
1216
+ {"timestamp_utc": "2026-04-11T21:55:26Z", "mode": "train", "global_step": 1193, "epoch": 0.04606889094840902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.387878787878789e-06, "num_tokens": 2569709.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012214966118335724, "sampling/sampling_logp_difference/max": 0.04612874984741211, "sampling/importance_sampling_ratio/min": 0.9549189805984497, "sampling/importance_sampling_ratio/mean": 1.000608205795288, "sampling/importance_sampling_ratio/max": 1.0443377494812012, "entropy": 0.009338034316897392, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0}
1217
+ {"timestamp_utc": "2026-04-11T21:55:32Z", "mode": "train", "global_step": 1194, "epoch": 0.04610750695088044, "loss": -0.0001, "grad_norm": 0.1992160677909851, "learning_rate": 6.384848484848485e-06, "num_tokens": 2572614.0, "completions/mean_length": 181.125, "completions/min_length": 181.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.125, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9984710216522217, "rewards/meter/std": 2.833232247212436e-05, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8320591449737549, "rewards/total_composite/std": 2.3610265998286195e-05, "reward": 0.8320591449737549, "reward_std": 2.3623691959073767e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009484349866397679, "sampling/sampling_logp_difference/max": 0.3491075038909912, "sampling/importance_sampling_ratio/min": 0.7053173184394836, "sampling/importance_sampling_ratio/mean": 0.9998201131820679, "sampling/importance_sampling_ratio/max": 1.209038496017456, "entropy": 0.006256086868233979, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8320591449737549, "reward_meter_mean": 0.9984710216522217, "reward_meter_std": 2.833232247212436e-05, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8320591449737549, "reward_total_composite_std": 2.3610265998286195e-05}
1218
+ {"timestamp_utc": "2026-04-11T21:55:38Z", "mode": "train", "global_step": 1195, "epoch": 0.04614612295335187, "loss": -0.0152, "grad_norm": 2.0341482162475586, "learning_rate": 6.381818181818182e-06, "num_tokens": 2575435.0, "completions/mean_length": 151.625, "completions/min_length": 142.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.625, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9975845813751221, "rewards/meter/std": 0.0008180320146493614, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7980676889419556, "rewards/total_composite/std": 0.0006544252391904593, "reward": 0.7980676889419556, "reward_std": 0.0006544221541844308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010841303505003452, "sampling/sampling_logp_difference/max": 0.9098663330078125, "sampling/importance_sampling_ratio/min": 0.40257805585861206, "sampling/importance_sampling_ratio/mean": 1.002147912979126, "sampling/importance_sampling_ratio/max": 1.647913932800293, "entropy": 0.02665393566712737, "clip_ratio/low_mean": 0.007669382495805621, "clip_ratio/low_min": 0.007669382495805621, "clip_ratio/high_mean": 0.004880740132648498, "clip_ratio/high_max": 0.004880740132648498, "clip_ratio/region_mean": 0.012550122628454119, "reward_total_mean": 0.7980676889419556, "reward_meter_mean": 0.9975845813751221, "reward_meter_std": 0.0008180320146493614, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7980676889419556, "reward_total_composite_std": 0.0006544252391904593}
1219
+ {"timestamp_utc": "2026-04-11T21:55:44Z", "mode": "train", "global_step": 1196, "epoch": 0.04618473895582329, "loss": -0.01, "grad_norm": 9.221901893615723, "learning_rate": 6.37878787878788e-06, "num_tokens": 2577632.0, "completions/mean_length": 103.625, "completions/min_length": 100.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9974844455718994, "rewards/meter/std": 0.0006991037516854703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974844455718994, "rewards/total_composite/std": 0.0006991037516854703, "reward": 0.9974844455718994, "reward_std": 0.0006991035188548267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015732292085886, "sampling/sampling_logp_difference/max": 2.215559244155884, "sampling/importance_sampling_ratio/min": 0.10909249633550644, "sampling/importance_sampling_ratio/mean": 0.9986835718154907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03731701336801052, "clip_ratio/low_mean": 0.014889341779053211, "clip_ratio/low_min": 0.014889341779053211, "clip_ratio/high_mean": 0.0035601977724581957, "clip_ratio/high_max": 0.0035601977724581957, "clip_ratio/region_mean": 0.018449539551511407, "reward_total_mean": 0.9974844455718994, "reward_meter_mean": 0.9974844455718994, "reward_meter_std": 0.0006991037516854703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974844455718994, "reward_total_composite_std": 0.0006991037516854703}
1220
+ {"timestamp_utc": "2026-04-11T21:55:48Z", "mode": "train", "global_step": 1197, "epoch": 0.046223354958294716, "loss": 0.1338, "grad_norm": 24.279144287109375, "learning_rate": 6.375757575757576e-06, "num_tokens": 2579110.0, "completions/mean_length": 35.75, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9846537113189697, "rewards/meter/std": 0.032167691737413406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9846537113189697, "rewards/total_composite/std": 0.032167691737413406, "reward": 0.9846537113189697, "reward_std": 0.03216767683625221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037709400057792664, "sampling/sampling_logp_difference/max": 3.0687551498413086, "sampling/importance_sampling_ratio/min": 0.04647897556424141, "sampling/importance_sampling_ratio/mean": 0.995938777923584, "sampling/importance_sampling_ratio/max": 1.4753588438034058, "entropy": 0.05178397847339511, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.018585751531645656, "clip_ratio/high_max": 0.018585751531645656, "clip_ratio/region_mean": 0.021245325915515423, "reward_total_mean": 0.9846537113189697, "reward_meter_mean": 0.9846537113189697, "reward_meter_std": 0.032167691737413406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9846537113189697, "reward_total_composite_std": 0.032167691737413406}
1221
+ {"timestamp_utc": "2026-04-11T21:55:58Z", "mode": "train", "global_step": 1198, "epoch": 0.04626197096076614, "loss": -0.0433, "grad_norm": 4.205373287200928, "learning_rate": 6.372727272727274e-06, "num_tokens": 2580935.0, "completions/mean_length": 128.125, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.28572082519531, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.042788296937942505, "rewards/meter/std": 0.09419418126344681, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.042788296937942505, "rewards/total_composite/std": 0.09419418126344681, "reward": 0.042788296937942505, "reward_std": 0.09419417381286621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06149071082472801, "sampling/sampling_logp_difference/max": 3.791663408279419, "sampling/importance_sampling_ratio/min": 0.022558048367500305, "sampling/importance_sampling_ratio/mean": 0.9972495436668396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17751457169651985, "clip_ratio/low_mean": 0.025734331109561026, "clip_ratio/low_min": 0.025734331109561026, "clip_ratio/high_mean": 0.001623376621864736, "clip_ratio/high_max": 0.001623376621864736, "clip_ratio/region_mean": 0.027357707731425762, "reward_total_mean": 0.042788296937942505, "reward_meter_mean": 0.042788296937942505, "reward_meter_std": 0.09419418126344681, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.042788296937942505, "reward_total_composite_std": 0.09419418126344681}
1222
+ {"timestamp_utc": "2026-04-11T21:56:04Z", "mode": "train", "global_step": 1199, "epoch": 0.046300586963237564, "loss": 0.0003, "grad_norm": 0.07112119346857071, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2583063.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984593391418457, "rewards/meter/std": 2.086405856971396e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984593391418457, "rewards/total_composite/std": 2.086405856971396e-06, "reward": 0.9984593391418457, "reward_std": 2.0953264083800605e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0007174843340180814, "sampling/sampling_logp_difference/max": 0.18243646621704102, "sampling/importance_sampling_ratio/min": 0.8332375884056091, "sampling/importance_sampling_ratio/mean": 1.0001569986343384, "sampling/importance_sampling_ratio/max": 1.0703777074813843, "entropy": 0.006159535580081865, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984593391418457, "reward_meter_mean": 0.9984593391418457, "reward_meter_std": 2.086405856971396e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984593391418457, "reward_total_composite_std": 2.086405856971396e-06}
1223
+ {"timestamp_utc": "2026-04-11T21:56:08Z", "mode": "train", "global_step": 1200, "epoch": 0.04633920296570899, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.366666666666668e-06, "num_tokens": 2584463.0, "completions/mean_length": 24.0, "completions/min_length": 24.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9885647892951965, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9885647892951965, "rewards/total_composite/std": 0.0, "reward": 0.9885647892951965, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006689532892778516, "sampling/sampling_logp_difference/max": 0.01897343248128891, "sampling/importance_sampling_ratio/min": 0.981205403804779, "sampling/importance_sampling_ratio/mean": 1.000182867050171, "sampling/importance_sampling_ratio/max": 1.0128138065338135, "entropy": 0.006478953931946307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9885647892951965, "reward_meter_mean": 0.9885647892951965, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9885647892951965, "reward_total_composite_std": 0.0}
1224
+ {"timestamp_utc": "2026-04-11T21:57:39Z", "mode": "eval", "global_step": 1200, "epoch": 0.04633920296570899, "eval_loss": NaN, "eval_runtime": 90.6551, "eval_samples_per_second": 1.147, "eval_steps_per_second": 0.143, "eval_num_tokens": 2584463.0, "eval_completions/mean_length": 250.20192307692307, "eval_completions/min_length": 58.92307692307692, "eval_completions/max_length": 487.9230769230769, "eval_completions/clipped_ratio": 0.16346153846153846, "eval_completions/mean_terminated_length": 198.65751765324518, "eval_completions/min_terminated_length": 58.92307692307692, "eval_completions/max_terminated_length": 423.6923076923077, "eval_rewards/meter/mean": 0.5284909640367215, "eval_rewards/meter/std": 0.49147624923632693, "eval_rewards/count_adherence/mean": 0.8193528331243075, "eval_rewards/count_adherence/std": 0.28436795794046843, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.4919216437981679, "eval_rewards/total_composite/std": 0.4640866976517897, "eval_reward": 0.4919216437981679, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.002731083811690601, "eval_sampling/sampling_logp_difference/max": 0.5404733006770794, "eval_sampling/importance_sampling_ratio/min": 0.6044848309113429, "eval_sampling/importance_sampling_ratio/mean": 1.0002888853733356, "eval_sampling/importance_sampling_ratio/max": 1.286704604442303, "eval_entropy": 0.01765880210754963, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4919216437981679, "eval_reward_meter_mean": 0.5284909640367215, "eval_reward_meter_std": 0.49147624923632693, "eval_reward_count_adherence_mean": 0.8193528331243075, "eval_reward_count_adherence_std": 0.28436795794046843, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.4919216437981679, "eval_reward_total_composite_std": 0.4640866976517897}
1225
+ {"timestamp_utc": "2026-04-11T21:57:46Z", "mode": "train", "global_step": 1201, "epoch": 0.04637781896818041, "loss": 0.0132, "grad_norm": 9.135161399841309, "learning_rate": 6.363636363636364e-06, "num_tokens": 2586236.0, "completions/mean_length": 52.625, "completions/min_length": 50.0, "completions/max_length": 55.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 52.625, "completions/min_terminated_length": 50.0, "completions/max_terminated_length": 55.0, "rewards/meter/mean": 0.6550425291061401, "rewards/meter/std": 0.43107059597969055, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.6550425291061401, "rewards/total_composite/std": 0.43107059597969055, "reward": 0.6550425291061401, "reward_std": 0.43107059597969055, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.03184327110648155, "sampling/sampling_logp_difference/max": 3.4988458156585693, "sampling/importance_sampling_ratio/min": 0.030232258141040802, "sampling/importance_sampling_ratio/mean": 0.9945924282073975, "sampling/importance_sampling_ratio/max": 1.7149361371994019, "entropy": 0.09002233669161797, "clip_ratio/low_mean": 0.007075471803545952, "clip_ratio/low_min": 0.007075471803545952, "clip_ratio/high_mean": 0.024252045433968306, "clip_ratio/high_max": 0.024252045433968306, "clip_ratio/region_mean": 0.03132751723751426, "reward_total_mean": 0.6550425291061401, "reward_meter_mean": 0.6550425291061401, "reward_meter_std": 0.43107059597969055, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.6550425291061401, "reward_total_composite_std": 0.43107059597969055}
plots/arabic_gate_chain.png CHANGED

Git LFS Details

  • SHA256: 19c09edec729c07e158531a2c3c19b2f5e3a10e2b3ba7f4077f3a8f1256bd393
  • Pointer size: 131 Bytes
  • Size of remote file: 108 kB

Git LFS Details

  • SHA256: ffeb3ac9d74957531486aed01d04bd436143dceb34c8f474ae3fe3bd59b85bf2
  • Pointer size: 131 Bytes
  • Size of remote file: 107 kB
plots/arabic_gate_run.png CHANGED

Git LFS Details

  • SHA256: 405ea6021b0ee1ff6c257c5ba356d7ea3f2fc654f5d190eca4c1e4f712d83296
  • Pointer size: 131 Bytes
  • Size of remote file: 107 kB

Git LFS Details

  • SHA256: e01ca2b1c6c926507d087e90c269138876e6aaabd1ec3c7a980143fb74066f59
  • Pointer size: 131 Bytes
  • Size of remote file: 106 kB
plots/chain_metrics.jsonl CHANGED
@@ -1170,3 +1170,54 @@
1170
  {"timestamp_utc": "2026-04-11T21:49:54Z", "mode": "train", "global_step": 1148, "epoch": 0.044331170837194935, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.524242424242425e-06, "num_tokens": 2477093.0, "completions/mean_length": 242.0, "completions/min_length": 242.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 242.0, "completions/min_terminated_length": 242.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970405697822571, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8724104762077332, "rewards/total_composite/std": 0.0, "reward": 0.8724104762077332, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.853495167684741e-05, "sampling/sampling_logp_difference/max": 0.01380898617208004, "sampling/importance_sampling_ratio/min": 0.9911676645278931, "sampling/importance_sampling_ratio/mean": 1.000036358833313, "sampling/importance_sampling_ratio/max": 1.0139048099517822, "entropy": 0.000526532585354289, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8724104762077332, "reward_meter_mean": 0.9970405697822571, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8724104762077332, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1171
  {"timestamp_utc": "2026-04-11T21:50:03Z", "mode": "train", "global_step": 1149, "epoch": 0.04436978683966636, "loss": -0.0126, "grad_norm": 0.35263416171073914, "learning_rate": 6.521212121212121e-06, "num_tokens": 2481621.0, "completions/mean_length": 372.0, "completions/min_length": 370.0, "completions/max_length": 386.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.0, "completions/min_terminated_length": 370.0, "completions/max_terminated_length": 386.0, "rewards/meter/mean": 0.9970118999481201, "rewards/meter/std": 1.4546116062774672e-06, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9243130683898926, "rewards/total_composite/std": 0.029373299330472946, "reward": 0.9243130683898926, "reward_std": 0.029373306781053543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017782310023903847, "sampling/sampling_logp_difference/max": 2.289862632751465, "sampling/importance_sampling_ratio/min": 0.10128037631511688, "sampling/importance_sampling_ratio/mean": 0.99934321641922, "sampling/importance_sampling_ratio/max": 1.0790009498596191, "entropy": 0.0008104511016426841, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00032383418874815106, "clip_ratio/high_max": 0.00032383418874815106, "clip_ratio/region_mean": 0.00032383418874815106, "reward_total_mean": 0.9243130683898926, "reward_meter_mean": 0.9970118999481201, "reward_meter_std": 1.4546116062774672e-06, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9243130683898926, "reward_total_composite_std": 0.029373299330472946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1172
  {"timestamp_utc": "2026-04-11T21:50:08Z", "mode": "train", "global_step": 1150, "epoch": 0.04440840284213778, "loss": 0.0522, "grad_norm": 5.2525129318237305, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2483289.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8777114748954773, "rewards/meter/std": 0.16049861907958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777114748954773, "rewards/total_composite/std": 0.16049861907958984, "reward": 0.8777114748954773, "reward_std": 0.16049860417842865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024218173697590828, "sampling/sampling_logp_difference/max": 5.00484561920166, "sampling/importance_sampling_ratio/min": 0.006705376319587231, "sampling/importance_sampling_ratio/mean": 0.9977035522460938, "sampling/importance_sampling_ratio/max": 1.5273561477661133, "entropy": 0.05149026960134506, "clip_ratio/low_mean": 0.004067460540682077, "clip_ratio/low_min": 0.004067460540682077, "clip_ratio/high_mean": 0.009015594609081745, "clip_ratio/high_max": 0.009015594609081745, "clip_ratio/region_mean": 0.013083055149763823, "reward_total_mean": 0.8777114748954773, "reward_meter_mean": 0.8777114748954773, "reward_meter_std": 0.16049861907958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777114748954773, "reward_total_composite_std": 0.16049861907958984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1170
  {"timestamp_utc": "2026-04-11T21:49:54Z", "mode": "train", "global_step": 1148, "epoch": 0.044331170837194935, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.524242424242425e-06, "num_tokens": 2477093.0, "completions/mean_length": 242.0, "completions/min_length": 242.0, "completions/max_length": 242.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 242.0, "completions/min_terminated_length": 242.0, "completions/max_terminated_length": 242.0, "rewards/meter/mean": 0.9970405697822571, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.875, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8724104762077332, "rewards/total_composite/std": 0.0, "reward": 0.8724104762077332, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 5.853495167684741e-05, "sampling/sampling_logp_difference/max": 0.01380898617208004, "sampling/importance_sampling_ratio/min": 0.9911676645278931, "sampling/importance_sampling_ratio/mean": 1.000036358833313, "sampling/importance_sampling_ratio/max": 1.0139048099517822, "entropy": 0.000526532585354289, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8724104762077332, "reward_meter_mean": 0.9970405697822571, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.875, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8724104762077332, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1171
  {"timestamp_utc": "2026-04-11T21:50:03Z", "mode": "train", "global_step": 1149, "epoch": 0.04436978683966636, "loss": -0.0126, "grad_norm": 0.35263416171073914, "learning_rate": 6.521212121212121e-06, "num_tokens": 2481621.0, "completions/mean_length": 372.0, "completions/min_length": 370.0, "completions/max_length": 386.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 372.0, "completions/min_terminated_length": 370.0, "completions/max_terminated_length": 386.0, "rewards/meter/mean": 0.9970118999481201, "rewards/meter/std": 1.4546116062774672e-06, "rewards/count_adherence/mean": 0.9270833730697632, "rewards/count_adherence/std": 0.029462777078151703, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9243130683898926, "rewards/total_composite/std": 0.029373299330472946, "reward": 0.9243130683898926, "reward_std": 0.029373306781053543, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0017782310023903847, "sampling/sampling_logp_difference/max": 2.289862632751465, "sampling/importance_sampling_ratio/min": 0.10128037631511688, "sampling/importance_sampling_ratio/mean": 0.99934321641922, "sampling/importance_sampling_ratio/max": 1.0790009498596191, "entropy": 0.0008104511016426841, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.00032383418874815106, "clip_ratio/high_max": 0.00032383418874815106, "clip_ratio/region_mean": 0.00032383418874815106, "reward_total_mean": 0.9243130683898926, "reward_meter_mean": 0.9970118999481201, "reward_meter_std": 1.4546116062774672e-06, "reward_count_adherence_mean": 0.9270833730697632, "reward_count_adherence_std": 0.029462777078151703, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9243130683898926, "reward_total_composite_std": 0.029373299330472946, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1172
  {"timestamp_utc": "2026-04-11T21:50:08Z", "mode": "train", "global_step": 1150, "epoch": 0.04440840284213778, "loss": 0.0522, "grad_norm": 5.2525129318237305, "learning_rate": 6.5181818181818195e-06, "num_tokens": 2483289.0, "completions/mean_length": 57.5, "completions/min_length": 54.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 57.5, "completions/min_terminated_length": 54.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.8777114748954773, "rewards/meter/std": 0.16049861907958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8777114748954773, "rewards/total_composite/std": 0.16049861907958984, "reward": 0.8777114748954773, "reward_std": 0.16049860417842865, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024218173697590828, "sampling/sampling_logp_difference/max": 5.00484561920166, "sampling/importance_sampling_ratio/min": 0.006705376319587231, "sampling/importance_sampling_ratio/mean": 0.9977035522460938, "sampling/importance_sampling_ratio/max": 1.5273561477661133, "entropy": 0.05149026960134506, "clip_ratio/low_mean": 0.004067460540682077, "clip_ratio/low_min": 0.004067460540682077, "clip_ratio/high_mean": 0.009015594609081745, "clip_ratio/high_max": 0.009015594609081745, "clip_ratio/region_mean": 0.013083055149763823, "reward_total_mean": 0.8777114748954773, "reward_meter_mean": 0.8777114748954773, "reward_meter_std": 0.16049861907958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8777114748954773, "reward_total_composite_std": 0.16049861907958984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1173
+ {"timestamp_utc": "2026-04-11T21:51:32Z", "mode": "eval", "global_step": 1150, "epoch": 0.04440840284213778, "eval_loss": NaN, "eval_runtime": 84.6125, "eval_samples_per_second": 1.229, "eval_steps_per_second": 0.154, "eval_num_tokens": 2483289.0, "eval_completions/mean_length": 216.47115384615384, "eval_completions/min_length": 60.76923076923077, "eval_completions/max_length": 447.0, "eval_completions/clipped_ratio": 0.057692307692307696, "eval_completions/mean_terminated_length": 197.60806157038763, "eval_completions/min_terminated_length": 60.76923076923077, "eval_completions/max_terminated_length": 396.38461538461536, "eval_rewards/meter/mean": 0.5331913347427661, "eval_rewards/meter/std": 0.45816060442190903, "eval_rewards/count_adherence/mean": 0.9469390053015488, "eval_rewards/count_adherence/std": 0.06932907207654072, "eval_rewards/arabic_clean/mean": 1.0, "eval_rewards/arabic_clean/std": 0.0, "eval_rewards/total_composite/mean": 0.511953374514213, "eval_rewards/total_composite/std": 0.440208015533594, "eval_reward": 0.511953374514213, "eval_reward_std": NaN, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.0024267814587801695, "eval_sampling/sampling_logp_difference/max": 0.5305701494216919, "eval_sampling/importance_sampling_ratio/min": 0.629830559858909, "eval_sampling/importance_sampling_ratio/mean": 1.0007386207580566, "eval_sampling/importance_sampling_ratio/max": 1.403554081916809, "eval_entropy": 0.017647026679836787, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.511953374514213, "eval_reward_meter_mean": 0.5331913347427661, "eval_reward_meter_std": 0.45816060442190903, "eval_reward_count_adherence_mean": 0.9469390053015488, "eval_reward_count_adherence_std": 0.06932907207654072, "eval_reward_arabic_clean_mean": 1.0, "eval_reward_arabic_clean_std": 0.0, "eval_reward_total_composite_mean": 0.511953374514213, "eval_reward_total_composite_std": 0.440208015533594, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1174
+ {"timestamp_utc": "2026-04-11T21:51:41Z", "mode": "train", "global_step": 1151, "epoch": 0.04444701884460921, "loss": 0.0504, "grad_norm": 3.7881219387054443, "learning_rate": 6.515151515151516e-06, "num_tokens": 2485639.0, "completions/mean_length": 122.75, "completions/min_length": 120.0, "completions/max_length": 136.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 122.75, "completions/min_terminated_length": 120.0, "completions/max_terminated_length": 136.0, "rewards/meter/mean": 0.888770341873169, "rewards/meter/std": 0.21289391815662384, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.888770341873169, "rewards/total_composite/std": 0.21289391815662384, "reward": 0.888770341873169, "reward_std": 0.21289391815662384, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.020240498706698418, "sampling/sampling_logp_difference/max": 6.1473388671875, "sampling/importance_sampling_ratio/min": 0.0021391669288277626, "sampling/importance_sampling_ratio/mean": 0.9987115263938904, "sampling/importance_sampling_ratio/max": 1.9864379167556763, "entropy": 0.013336000498384237, "clip_ratio/low_mean": 0.003968254197388887, "clip_ratio/low_min": 0.003968254197388887, "clip_ratio/high_mean": 0.006250000325962901, "clip_ratio/high_max": 0.006250000325962901, "clip_ratio/region_mean": 0.010218254523351789, "reward_total_mean": 0.888770341873169, "reward_meter_mean": 0.888770341873169, "reward_meter_std": 0.21289391815662384, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.888770341873169, "reward_total_composite_std": 0.21289391815662384, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1175
+ {"timestamp_utc": "2026-04-11T21:51:45Z", "mode": "train", "global_step": 1152, "epoch": 0.04448563484708063, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.512121212121213e-06, "num_tokens": 2487423.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 9.262973617296666e-05, "sampling/sampling_logp_difference/max": 0.002871689386665821, "sampling/importance_sampling_ratio/min": 0.9996728897094727, "sampling/importance_sampling_ratio/mean": 1.0000879764556885, "sampling/importance_sampling_ratio/max": 1.002875804901123, "entropy": 0.0009597428434062749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1176
+ {"timestamp_utc": "2026-04-11T21:51:50Z", "mode": "train", "global_step": 1153, "epoch": 0.044524250849552055, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.5090909090909095e-06, "num_tokens": 2489239.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00015291250019799918, "sampling/sampling_logp_difference/max": 0.014015945605933666, "sampling/importance_sampling_ratio/min": 0.9860818386077881, "sampling/importance_sampling_ratio/mean": 1.0000462532043457, "sampling/importance_sampling_ratio/max": 1.0045279264450073, "entropy": 0.00251059714355506, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1177
+ {"timestamp_utc": "2026-04-11T21:51:55Z", "mode": "train", "global_step": 1154, "epoch": 0.04456286685202348, "loss": 0.096, "grad_norm": 5.494170665740967, "learning_rate": 6.506060606060607e-06, "num_tokens": 2491171.0, "completions/mean_length": 65.5, "completions/min_length": 53.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.13582952320575714, "rewards/meter/std": 0.19803810119628906, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.13582952320575714, "rewards/total_composite/std": 0.19803810119628906, "reward": 0.13582952320575714, "reward_std": 0.19803808629512787, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041303541511297226, "sampling/sampling_logp_difference/max": 4.165966033935547, "sampling/importance_sampling_ratio/min": 0.015514720231294632, "sampling/importance_sampling_ratio/mean": 0.9954270124435425, "sampling/importance_sampling_ratio/max": 1.913017988204956, "entropy": 0.0860102130100131, "clip_ratio/low_mean": 0.013047155574895442, "clip_ratio/low_min": 0.013047155574895442, "clip_ratio/high_mean": 0.008954269345849752, "clip_ratio/high_max": 0.008954269345849752, "clip_ratio/region_mean": 0.022001424920745194, "reward_total_mean": 0.13582952320575714, "reward_meter_mean": 0.13582952320575714, "reward_meter_std": 0.19803810119628906, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.13582952320575714, "reward_total_composite_std": 0.19803810119628906, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1178
+ {"timestamp_utc": "2026-04-11T21:52:02Z", "mode": "train", "global_step": 1155, "epoch": 0.0446014828544949, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.503030303030303e-06, "num_tokens": 2493723.0, "completions/mean_length": 130.0, "completions/min_length": 130.0, "completions/max_length": 130.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.0, "completions/min_terminated_length": 130.0, "completions/max_terminated_length": 130.0, "rewards/meter/mean": 0.9970986843109131, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9970986843109131, "rewards/total_composite/std": 0.0, "reward": 0.9970986843109131, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001729379582684487, "sampling/sampling_logp_difference/max": 0.117275670170784, "sampling/importance_sampling_ratio/min": 0.8893400430679321, "sampling/importance_sampling_ratio/mean": 0.9999405145645142, "sampling/importance_sampling_ratio/max": 1.0045697689056396, "entropy": 0.0005436375031422358, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9970986843109131, "reward_meter_mean": 0.9970986843109131, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9970986843109131, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1179
+ {"timestamp_utc": "2026-04-11T21:52:12Z", "mode": "train", "global_step": 1156, "epoch": 0.04464009885696633, "loss": -0.0346, "grad_norm": 3.3614652156829834, "learning_rate": 6.5000000000000004e-06, "num_tokens": 2497463.0, "completions/mean_length": 304.5, "completions/min_length": 256.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 274.8571472167969, "completions/min_terminated_length": 256.0, "completions/max_terminated_length": 288.0, "rewards/meter/mean": 0.028450578451156616, "rewards/meter/std": 0.04791045933961868, "rewards/count_adherence/mean": 0.9285714626312256, "rewards/count_adherence/std": 0.2020305097103119, "rewards/arabic_clean/mean": 0.875, "rewards/arabic_clean/std": 0.3535533845424652, "rewards/total_composite/mean": 0.025792036205530167, "rewards/total_composite/std": 0.048944856971502304, "reward": 0.025792036205530167, "reward_std": 0.048944856971502304, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024456597864627838, "sampling/sampling_logp_difference/max": 4.192187309265137, "sampling/importance_sampling_ratio/min": 0.015113191679120064, "sampling/importance_sampling_ratio/mean": 1.001096487045288, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.04777472233399749, "clip_ratio/low_mean": 0.014450150192715228, "clip_ratio/low_min": 0.014450150192715228, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.016403275192715228, "reward_total_mean": 0.025792036205530167, "reward_meter_mean": 0.028450578451156616, "reward_meter_std": 0.04791045933961868, "reward_count_adherence_mean": 0.9285714626312256, "reward_count_adherence_std": 0.2020305097103119, "reward_arabic_clean_mean": 0.875, "reward_arabic_clean_std": 0.3535533845424652, "reward_total_composite_mean": 0.025792036205530167, "reward_total_composite_std": 0.048944856971502304, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1180
+ {"timestamp_utc": "2026-04-11T21:52:16Z", "mode": "train", "global_step": 1157, "epoch": 0.04467871485943775, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.496969696969697e-06, "num_tokens": 2499215.0, "completions/mean_length": 66.0, "completions/min_length": 66.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 66.0, "completions/min_terminated_length": 66.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.9972342252731323, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9972342252731323, "rewards/total_composite/std": 0.0, "reward": 0.9972342252731323, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0148831931874156, "sampling/sampling_logp_difference/max": 3.352174758911133, "sampling/importance_sampling_ratio/min": 0.03500813990831375, "sampling/importance_sampling_ratio/mean": 0.9938147664070129, "sampling/importance_sampling_ratio/max": 1.0112634897232056, "entropy": 0.006829831196228042, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9972342252731323, "reward_meter_mean": 0.9972342252731323, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9972342252731323, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1181
+ {"timestamp_utc": "2026-04-11T21:52:21Z", "mode": "train", "global_step": 1158, "epoch": 0.044717330861909176, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.493939393939395e-06, "num_tokens": 2501127.0, "completions/mean_length": 72.0, "completions/min_length": 72.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.0, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.9888595938682556, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888595938682556, "rewards/total_composite/std": 0.0, "reward": 0.9888595938682556, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00017435323388781399, "sampling/sampling_logp_difference/max": 0.0043810089118778706, "sampling/importance_sampling_ratio/min": 0.9968689680099487, "sampling/importance_sampling_ratio/mean": 1.000144600868225, "sampling/importance_sampling_ratio/max": 1.0043905973434448, "entropy": 0.0018330513557884842, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888595938682556, "reward_meter_mean": 0.9888595938682556, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888595938682556, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1182
+ {"timestamp_utc": "2026-04-11T21:52:26Z", "mode": "train", "global_step": 1159, "epoch": 0.0447559468643806, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.490909090909091e-06, "num_tokens": 2503135.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013290751667227596, "sampling/sampling_logp_difference/max": 0.006528853438794613, "sampling/importance_sampling_ratio/min": 0.9994356036186218, "sampling/importance_sampling_ratio/mean": 1.0001311302185059, "sampling/importance_sampling_ratio/max": 1.0065501928329468, "entropy": 0.0013790328812319785, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1183
+ {"timestamp_utc": "2026-04-11T21:52:31Z", "mode": "train", "global_step": 1160, "epoch": 0.044794562866852024, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.487878787878789e-06, "num_tokens": 2504759.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0001226484455401078, "sampling/sampling_logp_difference/max": 0.005975149571895599, "sampling/importance_sampling_ratio/min": 0.9942424893379211, "sampling/importance_sampling_ratio/mean": 1.0000890493392944, "sampling/importance_sampling_ratio/max": 1.005993127822876, "entropy": 0.001158960752945859, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1184
+ {"timestamp_utc": "2026-04-11T21:52:36Z", "mode": "train", "global_step": 1161, "epoch": 0.04483317886932345, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.484848484848485e-06, "num_tokens": 2506847.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00014731692499481142, "sampling/sampling_logp_difference/max": 0.005519423168152571, "sampling/importance_sampling_ratio/min": 0.9944958090782166, "sampling/importance_sampling_ratio/mean": 1.0000895261764526, "sampling/importance_sampling_ratio/max": 1.0041718482971191, "entropy": 0.0013596047574537806, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1185
+ {"timestamp_utc": "2026-04-11T21:52:41Z", "mode": "train", "global_step": 1162, "epoch": 0.04487179487179487, "loss": -0.0067, "grad_norm": 9.667710304260254, "learning_rate": 6.481818181818182e-06, "num_tokens": 2508562.0, "completions/mean_length": 54.375, "completions/min_length": 53.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.375, "completions/min_terminated_length": 53.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.9774544835090637, "rewards/meter/std": 0.007204634603112936, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9774544835090637, "rewards/total_composite/std": 0.007204634603112936, "reward": 0.9774544835090637, "reward_std": 0.007204628549516201, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.024649647995829582, "sampling/sampling_logp_difference/max": 3.8922629356384277, "sampling/importance_sampling_ratio/min": 0.02039913274347782, "sampling/importance_sampling_ratio/mean": 0.9978480339050293, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.028170868754386902, "clip_ratio/low_mean": 0.004629629664123058, "clip_ratio/low_min": 0.004629629664123058, "clip_ratio/high_mean": 0.004590633558109403, "clip_ratio/high_max": 0.004590633558109403, "clip_ratio/region_mean": 0.009220263222232461, "reward_total_mean": 0.9774544835090637, "reward_meter_mean": 0.9774544835090637, "reward_meter_std": 0.007204634603112936, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9774544835090637, "reward_total_composite_std": 0.007204634603112936, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1186
+ {"timestamp_utc": "2026-04-11T21:52:47Z", "mode": "train", "global_step": 1163, "epoch": 0.044910410874266296, "loss": -0.0003, "grad_norm": 0.0491514727473259, "learning_rate": 6.478787878787879e-06, "num_tokens": 2511042.0, "completions/mean_length": 145.0, "completions/min_length": 145.0, "completions/max_length": 145.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 145.0, "completions/min_terminated_length": 145.0, "completions/max_terminated_length": 145.0, "rewards/meter/mean": 0.9984663724899292, "rewards/meter/std": 3.918135462299688e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984663724899292, "rewards/total_composite/std": 3.918135462299688e-06, "reward": 0.9984663724899292, "reward_std": 3.918005404557334e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0012445234460756183, "sampling/sampling_logp_difference/max": 0.6710500717163086, "sampling/importance_sampling_ratio/min": 0.8294064402580261, "sampling/importance_sampling_ratio/mean": 1.000813603401184, "sampling/importance_sampling_ratio/max": 1.9562904834747314, "entropy": 0.006222540338058025, "clip_ratio/low_mean": 0.0008620689623057842, "clip_ratio/low_min": 0.0008620689623057842, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0008620689623057842, "reward_total_mean": 0.9984663724899292, "reward_meter_mean": 0.9984663724899292, "reward_meter_std": 3.918135462299688e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984663724899292, "reward_total_composite_std": 3.918135462299688e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1187
+ {"timestamp_utc": "2026-04-11T21:52:51Z", "mode": "train", "global_step": 1164, "epoch": 0.04494902687673772, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.475757575757576e-06, "num_tokens": 2512906.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 8.520409755874425e-05, "sampling/sampling_logp_difference/max": 0.0016170135932043195, "sampling/importance_sampling_ratio/min": 0.9988471865653992, "sampling/importance_sampling_ratio/mean": 1.000075340270996, "sampling/importance_sampling_ratio/max": 1.0016183853149414, "entropy": 0.0006748539672116749, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1188
+ {"timestamp_utc": "2026-04-11T21:52:56Z", "mode": "train", "global_step": 1165, "epoch": 0.044987642879209144, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.472727272727272e-06, "num_tokens": 2514546.0, "completions/mean_length": 62.0, "completions/min_length": 62.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.0, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.9959487915039062, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9959487915039062, "rewards/total_composite/std": 0.0, "reward": 0.9959487915039062, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00018705571710597724, "sampling/sampling_logp_difference/max": 0.003838915377855301, "sampling/importance_sampling_ratio/min": 0.996885359287262, "sampling/importance_sampling_ratio/mean": 1.0001657009124756, "sampling/importance_sampling_ratio/max": 1.003846287727356, "entropy": 0.0014549151237588376, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9959487915039062, "reward_meter_mean": 0.9959487915039062, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9959487915039062, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1189
+ {"timestamp_utc": "2026-04-11T21:53:01Z", "mode": "train", "global_step": 1166, "epoch": 0.04502625888168057, "loss": 0.0454, "grad_norm": 190.9034881591797, "learning_rate": 6.4696969696969705e-06, "num_tokens": 2516434.0, "completions/mean_length": 58.0, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.996626615524292, "rewards/meter/std": 0.001510909991338849, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.996626615524292, "rewards/total_composite/std": 0.001510909991338849, "reward": 0.996626615524292, "reward_std": 0.0015109025407582521, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.025698505342006683, "sampling/sampling_logp_difference/max": 1.1933188438415527, "sampling/importance_sampling_ratio/min": 0.3032132685184479, "sampling/importance_sampling_ratio/mean": 0.9995715618133545, "sampling/importance_sampling_ratio/max": 1.7860766649246216, "entropy": 0.059585667215287685, "clip_ratio/low_mean": 0.006657268386334181, "clip_ratio/low_min": 0.006657268386334181, "clip_ratio/high_mean": 0.008513708598911762, "clip_ratio/high_max": 0.008513708598911762, "clip_ratio/region_mean": 0.015170976985245943, "reward_total_mean": 0.996626615524292, "reward_meter_mean": 0.996626615524292, "reward_meter_std": 0.001510909991338849, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.996626615524292, "reward_total_composite_std": 0.001510909991338849, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1190
+ {"timestamp_utc": "2026-04-11T21:53:06Z", "mode": "train", "global_step": 1167, "epoch": 0.04506487488415199, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.466666666666667e-06, "num_tokens": 2518002.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0013065863167867064, "sampling/sampling_logp_difference/max": 0.18986795842647552, "sampling/importance_sampling_ratio/min": 0.8270683288574219, "sampling/importance_sampling_ratio/mean": 0.9993685483932495, "sampling/importance_sampling_ratio/max": 1.032031774520874, "entropy": 0.006112938834121451, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1191
+ {"timestamp_utc": "2026-04-11T21:53:11Z", "mode": "train", "global_step": 1168, "epoch": 0.04510349088662342, "loss": 0.0287, "grad_norm": 7.1253252029418945, "learning_rate": 6.463636363636364e-06, "num_tokens": 2520049.0, "completions/mean_length": 81.875, "completions/min_length": 80.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 81.875, "completions/min_terminated_length": 80.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.03959798067808151, "rewards/meter/std": 0.011728010140359402, "rewards/count_adherence/mean": 0.9583333730697632, "rewards/count_adherence/std": 0.117851123213768, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.038292575627565384, "rewards/total_composite/std": 0.01326083205640316, "reward": 0.038292575627565384, "reward_std": 0.01326083205640316, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.028142500668764114, "sampling/sampling_logp_difference/max": 4.7991814613342285, "sampling/importance_sampling_ratio/min": 0.008236486464738846, "sampling/importance_sampling_ratio/mean": 0.9953240752220154, "sampling/importance_sampling_ratio/max": 1.6731361150741577, "entropy": 0.03918551583774388, "clip_ratio/low_mean": 0.014004629920236766, "clip_ratio/low_min": 0.014004629920236766, "clip_ratio/high_mean": 0.004611280397512019, "clip_ratio/high_max": 0.004611280397512019, "clip_ratio/region_mean": 0.018615910317748785, "reward_total_mean": 0.038292575627565384, "reward_meter_mean": 0.03959798067808151, "reward_meter_std": 0.011728010140359402, "reward_count_adherence_mean": 0.9583333730697632, "reward_count_adherence_std": 0.117851123213768, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.038292575627565384, "reward_total_composite_std": 0.01326083205640316, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1192
+ {"timestamp_utc": "2026-04-11T21:53:15Z", "mode": "train", "global_step": 1169, "epoch": 0.04514210688909484, "loss": -0.044, "grad_norm": 4.53950834274292, "learning_rate": 6.460606060606061e-06, "num_tokens": 2521727.0, "completions/mean_length": 58.75, "completions/min_length": 55.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.75, "completions/min_terminated_length": 55.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9976530075073242, "rewards/meter/std": 0.0003900358860846609, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9976530075073242, "rewards/total_composite/std": 0.0003900358860846609, "reward": 0.9976530075073242, "reward_std": 0.00039003457641229033, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.013675752095878124, "sampling/sampling_logp_difference/max": 0.7795699834823608, "sampling/importance_sampling_ratio/min": 0.4586032032966614, "sampling/importance_sampling_ratio/mean": 0.9979230165481567, "sampling/importance_sampling_ratio/max": 1.4813544750213623, "entropy": 0.0419846111908555, "clip_ratio/low_mean": 0.006818181602284312, "clip_ratio/low_min": 0.006818181602284312, "clip_ratio/high_mean": 0.002016128972172737, "clip_ratio/high_max": 0.002016128972172737, "clip_ratio/region_mean": 0.00883431057445705, "reward_total_mean": 0.9976530075073242, "reward_meter_mean": 0.9976530075073242, "reward_meter_std": 0.0003900358860846609, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9976530075073242, "reward_total_composite_std": 0.0003900358860846609, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1193
+ {"timestamp_utc": "2026-04-11T21:53:26Z", "mode": "train", "global_step": 1170, "epoch": 0.045180722891566265, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.457575757575758e-06, "num_tokens": 2523215.0, "completions/mean_length": 512.0, "completions/min_length": 512.0, "completions/max_length": 512.0, "completions/clipped_ratio": 1.0, "completions/mean_terminated_length": 0.0, "completions/min_terminated_length": 0.0, "completions/max_terminated_length": 0.0, "rewards/meter/mean": 0.9969618916511536, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545387387275696, "rewards/total_composite/std": 0.0, "reward": 0.8545387387275696, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0, "sampling/sampling_logp_difference/max": 0.0, "sampling/importance_sampling_ratio/min": 0.0, "sampling/importance_sampling_ratio/mean": 0.0, "sampling/importance_sampling_ratio/max": 0.0, "entropy": 0.0, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545387387275696, "reward_meter_mean": 0.9969618916511536, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545387387275696, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1194
+ {"timestamp_utc": "2026-04-11T21:53:30Z", "mode": "train", "global_step": 1171, "epoch": 0.04521933889403769, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.454545454545456e-06, "num_tokens": 2524991.0, "completions/mean_length": 61.0, "completions/min_length": 61.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.0, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00013418152229860425, "sampling/sampling_logp_difference/max": 0.004571585915982723, "sampling/importance_sampling_ratio/min": 0.999534547328949, "sampling/importance_sampling_ratio/mean": 1.0001263618469238, "sampling/importance_sampling_ratio/max": 1.0045820474624634, "entropy": 0.0013301519793458283, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1195
+ {"timestamp_utc": "2026-04-11T21:53:36Z", "mode": "train", "global_step": 1172, "epoch": 0.04525795489650911, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.451515151515152e-06, "num_tokens": 2527167.0, "completions/mean_length": 96.0, "completions/min_length": 96.0, "completions/max_length": 96.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 96.0, "completions/min_terminated_length": 96.0, "completions/max_terminated_length": 96.0, "rewards/meter/mean": 0.9888964295387268, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9888964295387268, "rewards/total_composite/std": 0.0, "reward": 0.9888964295387268, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0002919211983680725, "sampling/sampling_logp_difference/max": 0.04608858376741409, "sampling/importance_sampling_ratio/min": 0.9549573659896851, "sampling/importance_sampling_ratio/mean": 1.00016450881958, "sampling/importance_sampling_ratio/max": 1.0190335512161255, "entropy": 0.00159282027016161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9888964295387268, "reward_meter_mean": 0.9888964295387268, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9888964295387268, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1196
+ {"timestamp_utc": "2026-04-11T21:53:41Z", "mode": "train", "global_step": 1173, "epoch": 0.04529657089898054, "loss": 0.0337, "grad_norm": 20.588041305541992, "learning_rate": 6.4484848484848496e-06, "num_tokens": 2528827.0, "completions/mean_length": 48.5, "completions/min_length": 48.0, "completions/max_length": 52.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 52.0, "rewards/meter/mean": 0.9464411735534668, "rewards/meter/std": 0.11976895481348038, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9464411735534668, "rewards/total_composite/std": 0.11976895481348038, "reward": 0.9464411735534668, "reward_std": 0.11976895481348038, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008157458156347275, "sampling/sampling_logp_difference/max": 1.4072527885437012, "sampling/importance_sampling_ratio/min": 0.244814932346344, "sampling/importance_sampling_ratio/mean": 1.0024529695510864, "sampling/importance_sampling_ratio/max": 1.9219340085983276, "entropy": 0.018341065326239914, "clip_ratio/low_mean": 0.004807692486792803, "clip_ratio/low_min": 0.004807692486792803, "clip_ratio/high_mean": 0.0026041667442768812, "clip_ratio/high_max": 0.0026041667442768812, "clip_ratio/region_mean": 0.007411859231069684, "reward_total_mean": 0.9464411735534668, "reward_meter_mean": 0.9464411735534668, "reward_meter_std": 0.11976895481348038, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9464411735534668, "reward_total_composite_std": 0.11976895481348038, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1197
+ {"timestamp_utc": "2026-04-11T21:53:45Z", "mode": "train", "global_step": 1174, "epoch": 0.04533518690145196, "loss": 0.032, "grad_norm": 12.414958953857422, "learning_rate": 6.445454545454546e-06, "num_tokens": 2530280.0, "completions/mean_length": 30.625, "completions/min_length": 30.0, "completions/max_length": 33.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 30.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 33.0, "rewards/meter/mean": 0.9770581126213074, "rewards/meter/std": 0.03885127976536751, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9770581126213074, "rewards/total_composite/std": 0.03885127976536751, "reward": 0.9770581126213074, "reward_std": 0.03885127976536751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00510073360055685, "sampling/sampling_logp_difference/max": 0.34801721572875977, "sampling/importance_sampling_ratio/min": 0.7204657793045044, "sampling/importance_sampling_ratio/mean": 1.0016231536865234, "sampling/importance_sampling_ratio/max": 1.4162566661834717, "entropy": 0.015783087466843426, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.004166666883975267, "clip_ratio/high_max": 0.004166666883975267, "clip_ratio/region_mean": 0.004166666883975267, "reward_total_mean": 0.9770581126213074, "reward_meter_mean": 0.9770581126213074, "reward_meter_std": 0.03885127976536751, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9770581126213074, "reward_total_composite_std": 0.03885127976536751, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1198
+ {"timestamp_utc": "2026-04-11T21:53:50Z", "mode": "train", "global_step": 1175, "epoch": 0.045373802903923385, "loss": 0.0157, "grad_norm": 10.53234577178955, "learning_rate": 6.442424242424243e-06, "num_tokens": 2531910.0, "completions/mean_length": 62.75, "completions/min_length": 62.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.75, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.98736572265625, "rewards/meter/std": 0.02987661585211754, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.98736572265625, "rewards/total_composite/std": 0.02987661585211754, "reward": 0.98736572265625, "reward_std": 0.02987661026418209, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.009396832436323166, "sampling/sampling_logp_difference/max": 1.0190973281860352, "sampling/importance_sampling_ratio/min": 0.36092060804367065, "sampling/importance_sampling_ratio/mean": 1.0028626918792725, "sampling/importance_sampling_ratio/max": 1.5600427389144897, "entropy": 0.03662474290467799, "clip_ratio/low_mean": 0.001953125, "clip_ratio/low_min": 0.001953125, "clip_ratio/high_mean": 0.0059843831695616245, "clip_ratio/high_max": 0.0059843831695616245, "clip_ratio/region_mean": 0.007937508169561625, "reward_total_mean": 0.98736572265625, "reward_meter_mean": 0.98736572265625, "reward_meter_std": 0.02987661585211754, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.98736572265625, "reward_total_composite_std": 0.02987661585211754, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1199
+ {"timestamp_utc": "2026-04-11T21:53:58Z", "mode": "train", "global_step": 1176, "epoch": 0.04541241890639481, "loss": 0.0004, "grad_norm": 0.008205018006265163, "learning_rate": 6.43939393939394e-06, "num_tokens": 2536102.0, "completions/mean_length": 307.0, "completions/min_length": 307.0, "completions/max_length": 307.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 307.0, "completions/min_terminated_length": 307.0, "completions/max_terminated_length": 307.0, "rewards/meter/mean": 0.9984632730484009, "rewards/meter/std": 1.9595856883825036e-06, "rewards/count_adherence/mean": 0.8888888955116272, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8875229358673096, "rewards/total_composite/std": 1.7389269260092988e-06, "reward": 0.8875229358673096, "reward_std": 1.7291217773163226e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00040480130701325834, "sampling/sampling_logp_difference/max": 0.38402700424194336, "sampling/importance_sampling_ratio/min": 0.681113064289093, "sampling/importance_sampling_ratio/mean": 0.9998529553413391, "sampling/importance_sampling_ratio/max": 1.0504670143127441, "entropy": 0.001436470149201341, "clip_ratio/low_mean": 0.0004071661096531898, "clip_ratio/low_min": 0.0004071661096531898, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0004071661096531898, "reward_total_mean": 0.8875229358673096, "reward_meter_mean": 0.9984632730484009, "reward_meter_std": 1.9595856883825036e-06, "reward_count_adherence_mean": 0.8888888955116272, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8875229358673096, "reward_total_composite_std": 1.7389269260092988e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1200
+ {"timestamp_utc": "2026-04-11T21:54:03Z", "mode": "train", "global_step": 1177, "epoch": 0.045451034908866234, "loss": -0.0054, "grad_norm": 3.3288590908050537, "learning_rate": 6.436363636363637e-06, "num_tokens": 2537728.0, "completions/mean_length": 56.25, "completions/min_length": 56.0, "completions/max_length": 58.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.25, "completions/min_terminated_length": 56.0, "completions/max_terminated_length": 58.0, "rewards/meter/mean": 0.9876357316970825, "rewards/meter/std": 0.00018677377374842763, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9876357316970825, "rewards/total_composite/std": 0.00018677377374842763, "reward": 0.9876357316970825, "reward_std": 0.00018675869796425104, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014386294409632683, "sampling/sampling_logp_difference/max": 2.6926071643829346, "sampling/importance_sampling_ratio/min": 0.0677042007446289, "sampling/importance_sampling_ratio/mean": 0.9979600310325623, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.014292308245785534, "clip_ratio/low_mean": 0.0022321429569274187, "clip_ratio/low_min": 0.0022321429569274187, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0022321429569274187, "reward_total_mean": 0.9876357316970825, "reward_meter_mean": 0.9876357316970825, "reward_meter_std": 0.00018677377374842763, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9876357316970825, "reward_total_composite_std": 0.00018677377374842763, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1201
+ {"timestamp_utc": "2026-04-11T21:54:08Z", "mode": "train", "global_step": 1178, "epoch": 0.04548965091133766, "loss": 0.0171, "grad_norm": 13.329391479492188, "learning_rate": 6.433333333333333e-06, "num_tokens": 2539308.0, "completions/mean_length": 48.5, "completions/min_length": 47.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.5, "completions/min_terminated_length": 47.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.9343859553337097, "rewards/meter/std": 0.11850643157958984, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9343859553337097, "rewards/total_composite/std": 0.11850643157958984, "reward": 0.9343859553337097, "reward_std": 0.11850643903017044, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.027483860030770302, "sampling/sampling_logp_difference/max": 1.6382932662963867, "sampling/importance_sampling_ratio/min": 0.19431141018867493, "sampling/importance_sampling_ratio/mean": 0.9949454069137573, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.0684116561897099, "clip_ratio/low_mean": 0.0049019609577953815, "clip_ratio/low_min": 0.0049019609577953815, "clip_ratio/high_mean": 0.01535926922224462, "clip_ratio/high_max": 0.01535926922224462, "clip_ratio/region_mean": 0.020261230180040002, "reward_total_mean": 0.9343859553337097, "reward_meter_mean": 0.9343859553337097, "reward_meter_std": 0.11850643157958984, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9343859553337097, "reward_total_composite_std": 0.11850643157958984, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1202
+ {"timestamp_utc": "2026-04-11T21:54:14Z", "mode": "train", "global_step": 1179, "epoch": 0.04552826691380908, "loss": -0.0044, "grad_norm": 4.12630033493042, "learning_rate": 6.430303030303031e-06, "num_tokens": 2541097.0, "completions/mean_length": 61.625, "completions/min_length": 61.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.9966377019882202, "rewards/meter/std": 0.0008516657399013638, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9966377019882202, "rewards/total_composite/std": 0.0008516657399013638, "reward": 0.9966377019882202, "reward_std": 0.0008516703965142369, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.012902844697237015, "sampling/sampling_logp_difference/max": 1.0336610078811646, "sampling/importance_sampling_ratio/min": 0.35570234060287476, "sampling/importance_sampling_ratio/mean": 0.995635449886322, "sampling/importance_sampling_ratio/max": 1.394187092781067, "entropy": 0.03527964395470917, "clip_ratio/low_mean": 0.008163669612258673, "clip_ratio/low_min": 0.008163669612258673, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.010147796710953116, "reward_total_mean": 0.9966377019882202, "reward_meter_mean": 0.9966377019882202, "reward_meter_std": 0.0008516657399013638, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9966377019882202, "reward_total_composite_std": 0.0008516657399013638, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1203
+ {"timestamp_utc": "2026-04-11T21:54:20Z", "mode": "train", "global_step": 1180, "epoch": 0.045566882916280506, "loss": -0.0, "grad_norm": 0.15215128660202026, "learning_rate": 6.427272727272728e-06, "num_tokens": 2544066.0, "completions/mean_length": 162.125, "completions/min_length": 162.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 162.125, "completions/min_terminated_length": 162.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.9969795942306519, "rewards/meter/std": 5.755153688369319e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969795942306519, "rewards/total_composite/std": 5.755153688369319e-05, "reward": 0.9969795942306519, "reward_std": 5.756055543315597e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.00256105768494308, "sampling/sampling_logp_difference/max": 1.255385160446167, "sampling/importance_sampling_ratio/min": 0.28496605157852173, "sampling/importance_sampling_ratio/mean": 0.9987585544586182, "sampling/importance_sampling_ratio/max": 1.2257132530212402, "entropy": 0.005278411292238161, "clip_ratio/low_mean": 0.0015432098880410194, "clip_ratio/low_min": 0.0015432098880410194, "clip_ratio/high_mean": 0.0007668711477890611, "clip_ratio/high_max": 0.0007668711477890611, "clip_ratio/region_mean": 0.0023100810358300805, "reward_total_mean": 0.9969795942306519, "reward_meter_mean": 0.9969795942306519, "reward_meter_std": 5.755153688369319e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969795942306519, "reward_total_composite_std": 5.755153688369319e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1204
+ {"timestamp_utc": "2026-04-11T21:54:25Z", "mode": "train", "global_step": 1181, "epoch": 0.04560549891875193, "loss": -0.0015, "grad_norm": 1.7918148040771484, "learning_rate": 6.424242424242425e-06, "num_tokens": 2545862.0, "completions/mean_length": 62.5, "completions/min_length": 62.0, "completions/max_length": 63.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 62.5, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 63.0, "rewards/meter/mean": 0.998000979423523, "rewards/meter/std": 5.412446989794262e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.998000979423523, "rewards/total_composite/std": 5.412446989794262e-05, "reward": 0.998000979423523, "reward_std": 5.414646147983149e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.008854641579091549, "sampling/sampling_logp_difference/max": 0.7983036041259766, "sampling/importance_sampling_ratio/min": 0.4500918388366699, "sampling/importance_sampling_ratio/mean": 0.9972570538520813, "sampling/importance_sampling_ratio/max": 1.2694493532180786, "entropy": 0.02104826516006142, "clip_ratio/low_mean": 0.002016128972172737, "clip_ratio/low_min": 0.002016128972172737, "clip_ratio/high_mean": 0.0019841270986944437, "clip_ratio/high_max": 0.0019841270986944437, "clip_ratio/region_mean": 0.004000256070867181, "reward_total_mean": 0.998000979423523, "reward_meter_mean": 0.998000979423523, "reward_meter_std": 5.412446989794262e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.998000979423523, "reward_total_composite_std": 5.412446989794262e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1205
+ {"timestamp_utc": "2026-04-11T21:54:31Z", "mode": "train", "global_step": 1182, "epoch": 0.045644114921223354, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.4212121212121215e-06, "num_tokens": 2548958.0, "completions/mean_length": 210.0, "completions/min_length": 210.0, "completions/max_length": 210.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 210.0, "completions/min_terminated_length": 210.0, "completions/max_terminated_length": 210.0, "rewards/meter/mean": 0.9969598650932312, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 0.8571428656578064, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8545370101928711, "rewards/total_composite/std": 0.0, "reward": 0.8545370101928711, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0003782480489462614, "sampling/sampling_logp_difference/max": 0.14583073556423187, "sampling/importance_sampling_ratio/min": 0.8643040657043457, "sampling/importance_sampling_ratio/mean": 0.9999819397926331, "sampling/importance_sampling_ratio/max": 1.0694695711135864, "entropy": 0.0029908385331509635, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8545370101928711, "reward_meter_mean": 0.9969598650932312, "reward_meter_std": 0.0, "reward_count_adherence_mean": 0.8571428656578064, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8545370101928711, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1206
+ {"timestamp_utc": "2026-04-11T21:54:36Z", "mode": "train", "global_step": 1183, "epoch": 0.04568273092369478, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.418181818181819e-06, "num_tokens": 2550854.0, "completions/mean_length": 73.0, "completions/min_length": 73.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 73.0, "completions/min_terminated_length": 73.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.9984769821166992, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984769821166992, "rewards/total_composite/std": 0.0, "reward": 0.9984769821166992, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0008001961396075785, "sampling/sampling_logp_difference/max": 0.0588974803686142, "sampling/importance_sampling_ratio/min": 0.9428033828735352, "sampling/importance_sampling_ratio/mean": 1.0005358457565308, "sampling/importance_sampling_ratio/max": 1.0240943431854248, "entropy": 0.007733863138128072, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984769821166992, "reward_meter_mean": 0.9984769821166992, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984769821166992, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1207
+ {"timestamp_utc": "2026-04-11T21:54:42Z", "mode": "train", "global_step": 1184, "epoch": 0.0457213469261662, "loss": 0.0258, "grad_norm": 5.115583896636963, "learning_rate": 6.415151515151515e-06, "num_tokens": 2553118.0, "completions/mean_length": 118.0, "completions/min_length": 107.0, "completions/max_length": 128.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 118.0, "completions/min_terminated_length": 107.0, "completions/max_terminated_length": 128.0, "rewards/meter/mean": 0.011638942174613476, "rewards/meter/std": 0.01774718053638935, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.011638942174613476, "rewards/total_composite/std": 0.01774718053638935, "reward": 0.011638942174613476, "reward_std": 0.01774718053638935, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0742255225777626, "sampling/sampling_logp_difference/max": 4.726832389831543, "sampling/importance_sampling_ratio/min": 0.008854473941028118, "sampling/importance_sampling_ratio/mean": 0.9967266321182251, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.15224116947501898, "clip_ratio/low_mean": 0.03850760939531028, "clip_ratio/low_min": 0.03850760939531028, "clip_ratio/high_mean": 0.011892712675035, "clip_ratio/high_max": 0.011892712675035, "clip_ratio/region_mean": 0.05040032207034528, "reward_total_mean": 0.011638942174613476, "reward_meter_mean": 0.011638942174613476, "reward_meter_std": 0.01774718053638935, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.011638942174613476, "reward_total_composite_std": 0.01774718053638935, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1208
+ {"timestamp_utc": "2026-04-11T21:54:46Z", "mode": "train", "global_step": 1185, "epoch": 0.045759962928637626, "loss": -0.0021, "grad_norm": 0.43357840180397034, "learning_rate": 6.412121212121213e-06, "num_tokens": 2554678.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985817670822144, "rewards/meter/std": 2.903921813413035e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985817670822144, "rewards/total_composite/std": 2.903921813413035e-05, "reward": 0.9985817670822144, "reward_std": 2.9045202609268017e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.004531952552497387, "sampling/sampling_logp_difference/max": 1.0694332122802734, "sampling/importance_sampling_ratio/min": 0.34320297837257385, "sampling/importance_sampling_ratio/mean": 0.9974938631057739, "sampling/importance_sampling_ratio/max": 1.0249570608139038, "entropy": 0.0018207905377494171, "clip_ratio/low_mean": 0.004032257944345474, "clip_ratio/low_min": 0.004032257944345474, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.004032257944345474, "reward_total_mean": 0.9985817670822144, "reward_meter_mean": 0.9985817670822144, "reward_meter_std": 2.903921813413035e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985817670822144, "reward_total_composite_std": 2.903921813413035e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1209
+ {"timestamp_utc": "2026-04-11T21:54:52Z", "mode": "train", "global_step": 1186, "epoch": 0.04579857893110905, "loss": -0.0023, "grad_norm": 1.6016870737075806, "learning_rate": 6.40909090909091e-06, "num_tokens": 2556862.0, "completions/mean_length": 98.0, "completions/min_length": 97.0, "completions/max_length": 99.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 98.0, "completions/min_terminated_length": 97.0, "completions/max_terminated_length": 99.0, "rewards/meter/mean": 0.9969850778579712, "rewards/meter/std": 9.93004723568447e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9969850778579712, "rewards/total_composite/std": 9.93004723568447e-05, "reward": 0.9969850778579712, "reward_std": 9.929558291332796e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0031935099977999926, "sampling/sampling_logp_difference/max": 0.7791270017623901, "sampling/importance_sampling_ratio/min": 0.4588063955307007, "sampling/importance_sampling_ratio/mean": 0.9980478882789612, "sampling/importance_sampling_ratio/max": 1.0727391242980957, "entropy": 0.011278128949925303, "clip_ratio/low_mean": 0.0012886597542092204, "clip_ratio/low_min": 0.0012886597542092204, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0012886597542092204, "reward_total_mean": 0.9969850778579712, "reward_meter_mean": 0.9969850778579712, "reward_meter_std": 9.93004723568447e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9969850778579712, "reward_total_composite_std": 9.93004723568447e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1210
+ {"timestamp_utc": "2026-04-11T21:54:57Z", "mode": "train", "global_step": 1187, "epoch": 0.045837194933580475, "loss": 0.0003, "grad_norm": 0.6999150514602661, "learning_rate": 6.406060606060607e-06, "num_tokens": 2559022.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984546899795532, "rewards/meter/std": 4.7120189265115187e-05, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984546899795532, "rewards/total_composite/std": 4.7120189265115187e-05, "reward": 0.9984546899795532, "reward_std": 4.7120178351178765e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0013951655710116029, "sampling/sampling_logp_difference/max": 0.15581762790679932, "sampling/importance_sampling_ratio/min": 0.8960135579109192, "sampling/importance_sampling_ratio/mean": 1.000537633895874, "sampling/importance_sampling_ratio/max": 1.168613076210022, "entropy": 0.010229390056338161, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0011467889416962862, "clip_ratio/high_max": 0.0011467889416962862, "clip_ratio/region_mean": 0.0011467889416962862, "reward_total_mean": 0.9984546899795532, "reward_meter_mean": 0.9984546899795532, "reward_meter_std": 4.7120189265115187e-05, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984546899795532, "reward_total_composite_std": 4.7120189265115187e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1211
+ {"timestamp_utc": "2026-04-11T21:55:02Z", "mode": "train", "global_step": 1188, "epoch": 0.0458758109360519, "loss": 0.0782, "grad_norm": 3.00974702835083, "learning_rate": 6.403030303030303e-06, "num_tokens": 2560807.0, "completions/mean_length": 75.125, "completions/min_length": 72.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 75.125, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.8669778108596802, "rewards/meter/std": 0.342582643032074, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8669778108596802, "rewards/total_composite/std": 0.342582643032074, "reward": 0.8669778108596802, "reward_std": 0.342582643032074, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.014270931482315063, "sampling/sampling_logp_difference/max": 3.183518409729004, "sampling/importance_sampling_ratio/min": 0.04143959656357765, "sampling/importance_sampling_ratio/mean": 1.000046730041504, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.02158025815151632, "clip_ratio/low_mean": 0.0013736264081671834, "clip_ratio/low_min": 0.0013736264081671834, "clip_ratio/high_mean": 0.006677350495010614, "clip_ratio/high_max": 0.006677350495010614, "clip_ratio/region_mean": 0.008050976903177798, "reward_total_mean": 0.8669778108596802, "reward_meter_mean": 0.8669778108596802, "reward_meter_std": 0.342582643032074, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8669778108596802, "reward_total_composite_std": 0.342582643032074, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1212
+ {"timestamp_utc": "2026-04-11T21:55:07Z", "mode": "train", "global_step": 1189, "epoch": 0.04591442693852332, "loss": 0.0287, "grad_norm": 7.2858405113220215, "learning_rate": 6.4000000000000006e-06, "num_tokens": 2562683.0, "completions/mean_length": 65.5, "completions/min_length": 63.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.4144327640533447, "rewards/meter/std": 0.30921247601509094, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.4144327640533447, "rewards/total_composite/std": 0.30921247601509094, "reward": 0.4144327640533447, "reward_std": 0.30921247601509094, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.035145364701747894, "sampling/sampling_logp_difference/max": 2.764584541320801, "sampling/importance_sampling_ratio/min": 0.06300227344036102, "sampling/importance_sampling_ratio/mean": 1.0029542446136475, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.10702869668602943, "clip_ratio/low_mean": 0.022366520133800805, "clip_ratio/low_min": 0.022366520133800805, "clip_ratio/high_mean": 0.001953125, "clip_ratio/high_max": 0.001953125, "clip_ratio/region_mean": 0.024319645133800805, "reward_total_mean": 0.4144327640533447, "reward_meter_mean": 0.4144327640533447, "reward_meter_std": 0.30921247601509094, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.4144327640533447, "reward_total_composite_std": 0.30921247601509094, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1213
+ {"timestamp_utc": "2026-04-11T21:55:11Z", "mode": "train", "global_step": 1190, "epoch": 0.04595304294099475, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.396969696969697e-06, "num_tokens": 2564211.0, "completions/mean_length": 31.0, "completions/min_length": 31.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 31.0, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.00010159891098737717, "sampling/sampling_logp_difference/max": 0.0022197323851287365, "sampling/importance_sampling_ratio/min": 0.9999852180480957, "sampling/importance_sampling_ratio/mean": 1.0001014471054077, "sampling/importance_sampling_ratio/max": 1.0022221803665161, "entropy": 0.0008445165294688195, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1214
+ {"timestamp_utc": "2026-04-11T21:55:16Z", "mode": "train", "global_step": 1191, "epoch": 0.04599165894346617, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.393939393939394e-06, "num_tokens": 2566107.0, "completions/mean_length": 91.0, "completions/min_length": 91.0, "completions/max_length": 91.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 91.0, "completions/min_terminated_length": 91.0, "completions/max_terminated_length": 91.0, "rewards/meter/mean": 0.9985920786857605, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9985920786857605, "rewards/total_composite/std": 0.0, "reward": 0.9985920786857605, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 6.144684448372573e-05, "sampling/sampling_logp_difference/max": 0.001986202783882618, "sampling/importance_sampling_ratio/min": 0.999610185623169, "sampling/importance_sampling_ratio/mean": 1.0000598430633545, "sampling/importance_sampling_ratio/max": 1.001988172531128, "entropy": 0.00044536186032928526, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9985920786857605, "reward_meter_mean": 0.9985920786857605, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9985920786857605, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1215
+ {"timestamp_utc": "2026-04-11T21:55:21Z", "mode": "train", "global_step": 1192, "epoch": 0.046030274945937595, "loss": 0.0119, "grad_norm": 4.474715232849121, "learning_rate": 6.390909090909091e-06, "num_tokens": 2568053.0, "completions/mean_length": 64.25, "completions/min_length": 62.0, "completions/max_length": 66.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.25, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 66.0, "rewards/meter/mean": 0.687461256980896, "rewards/meter/std": 0.23647956550121307, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.687461256980896, "rewards/total_composite/std": 0.23647956550121307, "reward": 0.687461256980896, "reward_std": 0.23647956550121307, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.021633058786392212, "sampling/sampling_logp_difference/max": 0.8202657699584961, "sampling/importance_sampling_ratio/min": 0.4403146207332611, "sampling/importance_sampling_ratio/mean": 0.998434841632843, "sampling/importance_sampling_ratio/max": 1.4843707084655762, "entropy": 0.09414011146873236, "clip_ratio/low_mean": 0.005769230774603784, "clip_ratio/low_min": 0.005769230774603784, "clip_ratio/high_mean": 0.021223400719463825, "clip_ratio/high_max": 0.021223400719463825, "clip_ratio/region_mean": 0.02699263149406761, "reward_total_mean": 0.687461256980896, "reward_meter_mean": 0.687461256980896, "reward_meter_std": 0.23647956550121307, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.687461256980896, "reward_total_composite_std": 0.23647956550121307, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1216
+ {"timestamp_utc": "2026-04-11T21:55:26Z", "mode": "train", "global_step": 1193, "epoch": 0.04606889094840902, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.387878787878789e-06, "num_tokens": 2569709.0, "completions/mean_length": 48.0, "completions/min_length": 48.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.0, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.9887858629226685, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9887858629226685, "rewards/total_composite/std": 0.0, "reward": 0.9887858629226685, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0012214966118335724, "sampling/sampling_logp_difference/max": 0.04612874984741211, "sampling/importance_sampling_ratio/min": 0.9549189805984497, "sampling/importance_sampling_ratio/mean": 1.000608205795288, "sampling/importance_sampling_ratio/max": 1.0443377494812012, "entropy": 0.009338034316897392, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9887858629226685, "reward_meter_mean": 0.9887858629226685, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9887858629226685, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1217
+ {"timestamp_utc": "2026-04-11T21:55:32Z", "mode": "train", "global_step": 1194, "epoch": 0.04610750695088044, "loss": -0.0001, "grad_norm": 0.1992160677909851, "learning_rate": 6.384848484848485e-06, "num_tokens": 2572614.0, "completions/mean_length": 181.125, "completions/min_length": 181.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 181.125, "completions/min_terminated_length": 181.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.9984710216522217, "rewards/meter/std": 2.833232247212436e-05, "rewards/count_adherence/mean": 0.8333333134651184, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.8320591449737549, "rewards/total_composite/std": 2.3610265998286195e-05, "reward": 0.8320591449737549, "reward_std": 2.3623691959073767e-05, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0009484349866397679, "sampling/sampling_logp_difference/max": 0.3491075038909912, "sampling/importance_sampling_ratio/min": 0.7053173184394836, "sampling/importance_sampling_ratio/mean": 0.9998201131820679, "sampling/importance_sampling_ratio/max": 1.209038496017456, "entropy": 0.006256086868233979, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.8320591449737549, "reward_meter_mean": 0.9984710216522217, "reward_meter_std": 2.833232247212436e-05, "reward_count_adherence_mean": 0.8333333134651184, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.8320591449737549, "reward_total_composite_std": 2.3610265998286195e-05, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1218
+ {"timestamp_utc": "2026-04-11T21:55:38Z", "mode": "train", "global_step": 1195, "epoch": 0.04614612295335187, "loss": -0.0152, "grad_norm": 2.0341482162475586, "learning_rate": 6.381818181818182e-06, "num_tokens": 2575435.0, "completions/mean_length": 151.625, "completions/min_length": 142.0, "completions/max_length": 154.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 151.625, "completions/min_terminated_length": 142.0, "completions/max_terminated_length": 154.0, "rewards/meter/mean": 0.9975845813751221, "rewards/meter/std": 0.0008180320146493614, "rewards/count_adherence/mean": 0.800000011920929, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.7980676889419556, "rewards/total_composite/std": 0.0006544252391904593, "reward": 0.7980676889419556, "reward_std": 0.0006544221541844308, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.010841303505003452, "sampling/sampling_logp_difference/max": 0.9098663330078125, "sampling/importance_sampling_ratio/min": 0.40257805585861206, "sampling/importance_sampling_ratio/mean": 1.002147912979126, "sampling/importance_sampling_ratio/max": 1.647913932800293, "entropy": 0.02665393566712737, "clip_ratio/low_mean": 0.007669382495805621, "clip_ratio/low_min": 0.007669382495805621, "clip_ratio/high_mean": 0.004880740132648498, "clip_ratio/high_max": 0.004880740132648498, "clip_ratio/region_mean": 0.012550122628454119, "reward_total_mean": 0.7980676889419556, "reward_meter_mean": 0.9975845813751221, "reward_meter_std": 0.0008180320146493614, "reward_count_adherence_mean": 0.800000011920929, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.7980676889419556, "reward_total_composite_std": 0.0006544252391904593, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1219
+ {"timestamp_utc": "2026-04-11T21:55:44Z", "mode": "train", "global_step": 1196, "epoch": 0.04618473895582329, "loss": -0.01, "grad_norm": 9.221901893615723, "learning_rate": 6.37878787878788e-06, "num_tokens": 2577632.0, "completions/mean_length": 103.625, "completions/min_length": 100.0, "completions/max_length": 106.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 103.625, "completions/min_terminated_length": 100.0, "completions/max_terminated_length": 106.0, "rewards/meter/mean": 0.9974844455718994, "rewards/meter/std": 0.0006991037516854703, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9974844455718994, "rewards/total_composite/std": 0.0006991037516854703, "reward": 0.9974844455718994, "reward_std": 0.0006991035188548267, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.015732292085886, "sampling/sampling_logp_difference/max": 2.215559244155884, "sampling/importance_sampling_ratio/min": 0.10909249633550644, "sampling/importance_sampling_ratio/mean": 0.9986835718154907, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.03731701336801052, "clip_ratio/low_mean": 0.014889341779053211, "clip_ratio/low_min": 0.014889341779053211, "clip_ratio/high_mean": 0.0035601977724581957, "clip_ratio/high_max": 0.0035601977724581957, "clip_ratio/region_mean": 0.018449539551511407, "reward_total_mean": 0.9974844455718994, "reward_meter_mean": 0.9974844455718994, "reward_meter_std": 0.0006991037516854703, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9974844455718994, "reward_total_composite_std": 0.0006991037516854703, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1220
+ {"timestamp_utc": "2026-04-11T21:55:48Z", "mode": "train", "global_step": 1197, "epoch": 0.046223354958294716, "loss": 0.1338, "grad_norm": 24.279144287109375, "learning_rate": 6.375757575757576e-06, "num_tokens": 2579110.0, "completions/mean_length": 35.75, "completions/min_length": 31.0, "completions/max_length": 47.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 47.0, "rewards/meter/mean": 0.9846537113189697, "rewards/meter/std": 0.032167691737413406, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9846537113189697, "rewards/total_composite/std": 0.032167691737413406, "reward": 0.9846537113189697, "reward_std": 0.03216767683625221, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.037709400057792664, "sampling/sampling_logp_difference/max": 3.0687551498413086, "sampling/importance_sampling_ratio/min": 0.04647897556424141, "sampling/importance_sampling_ratio/mean": 0.995938777923584, "sampling/importance_sampling_ratio/max": 1.4753588438034058, "entropy": 0.05178397847339511, "clip_ratio/low_mean": 0.002659574383869767, "clip_ratio/low_min": 0.002659574383869767, "clip_ratio/high_mean": 0.018585751531645656, "clip_ratio/high_max": 0.018585751531645656, "clip_ratio/region_mean": 0.021245325915515423, "reward_total_mean": 0.9846537113189697, "reward_meter_mean": 0.9846537113189697, "reward_meter_std": 0.032167691737413406, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9846537113189697, "reward_total_composite_std": 0.032167691737413406, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1221
+ {"timestamp_utc": "2026-04-11T21:55:58Z", "mode": "train", "global_step": 1198, "epoch": 0.04626197096076614, "loss": -0.0433, "grad_norm": 4.205373287200928, "learning_rate": 6.372727272727274e-06, "num_tokens": 2580935.0, "completions/mean_length": 128.125, "completions/min_length": 23.0, "completions/max_length": 512.0, "completions/clipped_ratio": 0.125, "completions/mean_terminated_length": 73.28572082519531, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 87.0, "rewards/meter/mean": 0.042788296937942505, "rewards/meter/std": 0.09419418126344681, "rewards/count_adherence/mean": 0.75, "rewards/count_adherence/std": 0.4629100561141968, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.042788296937942505, "rewards/total_composite/std": 0.09419418126344681, "reward": 0.042788296937942505, "reward_std": 0.09419417381286621, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.06149071082472801, "sampling/sampling_logp_difference/max": 3.791663408279419, "sampling/importance_sampling_ratio/min": 0.022558048367500305, "sampling/importance_sampling_ratio/mean": 0.9972495436668396, "sampling/importance_sampling_ratio/max": 2.0, "entropy": 0.17751457169651985, "clip_ratio/low_mean": 0.025734331109561026, "clip_ratio/low_min": 0.025734331109561026, "clip_ratio/high_mean": 0.001623376621864736, "clip_ratio/high_max": 0.001623376621864736, "clip_ratio/region_mean": 0.027357707731425762, "reward_total_mean": 0.042788296937942505, "reward_meter_mean": 0.042788296937942505, "reward_meter_std": 0.09419418126344681, "reward_count_adherence_mean": 0.75, "reward_count_adherence_std": 0.4629100561141968, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.042788296937942505, "reward_total_composite_std": 0.09419418126344681, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1222
+ {"timestamp_utc": "2026-04-11T21:56:04Z", "mode": "train", "global_step": 1199, "epoch": 0.046300586963237564, "loss": 0.0003, "grad_norm": 0.07112119346857071, "learning_rate": 6.3696969696969706e-06, "num_tokens": 2583063.0, "completions/mean_length": 109.0, "completions/min_length": 109.0, "completions/max_length": 109.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 109.0, "completions/min_terminated_length": 109.0, "completions/max_terminated_length": 109.0, "rewards/meter/mean": 0.9984593391418457, "rewards/meter/std": 2.086405856971396e-06, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9984593391418457, "rewards/total_composite/std": 2.086405856971396e-06, "reward": 0.9984593391418457, "reward_std": 2.0953264083800605e-06, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.0007174843340180814, "sampling/sampling_logp_difference/max": 0.18243646621704102, "sampling/importance_sampling_ratio/min": 0.8332375884056091, "sampling/importance_sampling_ratio/mean": 1.0001569986343384, "sampling/importance_sampling_ratio/max": 1.0703777074813843, "entropy": 0.006159535580081865, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9984593391418457, "reward_meter_mean": 0.9984593391418457, "reward_meter_std": 2.086405856971396e-06, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9984593391418457, "reward_total_composite_std": 2.086405856971396e-06, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
1223
+ {"timestamp_utc": "2026-04-11T21:56:08Z", "mode": "train", "global_step": 1200, "epoch": 0.04633920296570899, "loss": 0.0, "grad_norm": 0.0, "learning_rate": 6.366666666666668e-06, "num_tokens": 2584463.0, "completions/mean_length": 24.0, "completions/min_length": 24.0, "completions/max_length": 24.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.0, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 24.0, "rewards/meter/mean": 0.9885647892951965, "rewards/meter/std": 0.0, "rewards/count_adherence/mean": 1.0, "rewards/count_adherence/std": 0.0, "rewards/arabic_clean/mean": 1.0, "rewards/arabic_clean/std": 0.0, "rewards/total_composite/mean": 0.9885647892951965, "rewards/total_composite/std": 0.0, "reward": 0.9885647892951965, "reward_std": 0.0, "frac_reward_zero_std": 1.0, "sampling/sampling_logp_difference/mean": 0.0006689532892778516, "sampling/sampling_logp_difference/max": 0.01897343248128891, "sampling/importance_sampling_ratio/min": 0.981205403804779, "sampling/importance_sampling_ratio/mean": 1.000182867050171, "sampling/importance_sampling_ratio/max": 1.0128138065338135, "entropy": 0.006478953931946307, "clip_ratio/low_mean": 0.0, "clip_ratio/low_min": 0.0, "clip_ratio/high_mean": 0.0, "clip_ratio/high_max": 0.0, "clip_ratio/region_mean": 0.0, "reward_total_mean": 0.9885647892951965, "reward_meter_mean": 0.9885647892951965, "reward_meter_std": 0.0, "reward_count_adherence_mean": 1.0, "reward_count_adherence_std": 0.0, "reward_arabic_clean_mean": 1.0, "reward_arabic_clean_std": 0.0, "reward_total_composite_mean": 0.9885647892951965, "reward_total_composite_std": 0.0, "run_id": "shaer_grpo_20260411_192107", "run_sequence_index": 0}
plots/meter_by_meter_chain.png CHANGED

Git LFS Details

  • SHA256: 868cea19592cbab90ef5742e459bfc6a128865af7351aafef836bd2a19015dbb
  • Pointer size: 131 Bytes
  • Size of remote file: 916 kB

Git LFS Details

  • SHA256: f58c910674e9e479d0eb58c4f9afc1b86da4b301beca1c2658f653082525d05c
  • Pointer size: 131 Bytes
  • Size of remote file: 931 kB
plots/meter_by_meter_run.png CHANGED

Git LFS Details

  • SHA256: e0552679313ce0c50a88097f941ddd16274ab3c5e88880f5e8a3e7a6032c795f
  • Pointer size: 131 Bytes
  • Size of remote file: 916 kB

Git LFS Details

  • SHA256: be55cd7e956ca3925f6658a305c38e79618c717e7a2f2d0b34b7d517bdd30a1f
  • Pointer size: 131 Bytes
  • Size of remote file: 930 kB
plots/reward_panels_eval_chain.png CHANGED

Git LFS Details

  • SHA256: 977dc9fd4faff1143103a70ce4da146439800f8387064b03f30f88872cd3106c
  • Pointer size: 131 Bytes
  • Size of remote file: 265 kB

Git LFS Details

  • SHA256: a79a5f40cbf5ea8e87e7870eadec6704bb2553b25513b62f4fcd93543f2b0e37
  • Pointer size: 131 Bytes
  • Size of remote file: 270 kB
plots/reward_panels_eval_run.png CHANGED

Git LFS Details

  • SHA256: e8bb91106b695795897747cb4d722bd42785e1de04dac07dd179b2363f6bb6e1
  • Pointer size: 131 Bytes
  • Size of remote file: 265 kB

Git LFS Details

  • SHA256: eb8d06be85d05a0227ecccb980dbd2b50c5c13fe83c48543169f573a423a7494
  • Pointer size: 131 Bytes
  • Size of remote file: 269 kB
plots/reward_panels_train_chain.png CHANGED

Git LFS Details

  • SHA256: 0d8080733f123180b636c1b3ddcc119deaf8b5308f15711b25bce4e508f41743
  • Pointer size: 131 Bytes
  • Size of remote file: 502 kB

Git LFS Details

  • SHA256: 5f0b44612270d6d497edb0c4f5bbb10f90e962ba06ed755a1d17510b010d17b4
  • Pointer size: 131 Bytes
  • Size of remote file: 501 kB
plots/reward_panels_train_run.png CHANGED

Git LFS Details

  • SHA256: 9d9292665a4624215952b31abc389b827f2a4b0e65ae10613a10cb24654f6b4e
  • Pointer size: 131 Bytes
  • Size of remote file: 502 kB

Git LFS Details

  • SHA256: 8a4306676ea966acdcbd26700fe61d58668741c2a92264e13f83460ec4032d80
  • Pointer size: 131 Bytes
  • Size of remote file: 500 kB
plotter.log CHANGED
@@ -2038,3 +2038,103 @@
2038
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2039
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2040
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2038
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2039
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2040
  [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2041
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2042
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2043
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2044
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2045
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2046
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2047
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2048
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2049
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2050
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2051
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2052
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2053
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2054
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2055
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2056
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2057
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2058
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2059
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2060
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2061
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2062
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2063
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2064
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2065
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2066
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2067
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2068
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2069
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2070
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2071
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2072
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2073
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2074
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2075
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2076
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2077
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2078
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2079
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2080
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2081
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2082
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2083
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2084
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2085
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2086
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2087
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2088
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2089
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2090
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2091
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2092
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2093
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2094
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2095
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2096
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2097
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2098
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2099
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2100
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2101
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2102
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2103
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2104
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2105
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2106
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2107
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2108
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2109
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2110
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2111
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2112
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2113
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2114
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2115
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2116
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2117
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2118
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2119
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2120
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2121
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2122
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2123
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2124
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2125
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2126
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2127
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2128
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2129
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2130
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
2131
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_run.png
2132
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_run.png
2133
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_train_chain.png
2134
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/reward_panels_eval_chain.png
2135
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_run.png
2136
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/kl_chain.png
2137
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_run.png
2138
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/arabic_gate_chain.png
2139
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_run.png
2140
+ [plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_192107/plots/meter_by_meter_chain.png
reward_arabic_clean_debug.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:2fd95bcc64ded28dcf931ae11388b98da805ae3bebffabcc37cee007667e4fd3
3
- size 43242649
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c9bcfe5e8589c4da728a5e51949ea6022015bed2983a7bfe5a0ea31511718c0b
3
+ size 45012913
reward_count_adherence_debug.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a42257d1ea09eb4567600f37b3da2be9bc2488cf13b0e3020fc2fcceb13c52e3
3
- size 43267449
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1d2cd74568ae88940c6578f1ca07c01952387d6b7051523e80d37cebc18646d8
3
+ size 45038728
reward_meter_debug.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e0e979be3c408d230ac8a96c957e6cf4a1c607679a0c1b5a9ec8d052933e1449
3
- size 43418595
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a250500d85b31423cd7d43302f7bcaa62ecc8f843d69a4940e68836046bb0527
3
+ size 45196339
reward_total_composite_debug.jsonl CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bbb12c0cbeb3e8b117d7cab7bdf3e0170f3af9b62378eceef0cc96e019316bd9
3
- size 43412202
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2e03f38d091f6443e7c105540d5a3387d4ec29d3b26c7b8160e48cbb779f093d
3
+ size 45189939
train.log CHANGED
@@ -1183,3 +1183,54 @@
1183
  2026-04-11 21:50:08,134 | INFO | train_grpo_train | metrics_logged mode=train step=1150
1184
  2026-04-11 21:51:32,807 | INFO | train_grpo_train | metrics_logged mode=eval step=1150
1185
  2026-04-11 21:51:41,218 | INFO | train_grpo_train | metrics_logged mode=train step=1151
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1183
  2026-04-11 21:50:08,134 | INFO | train_grpo_train | metrics_logged mode=train step=1150
1184
  2026-04-11 21:51:32,807 | INFO | train_grpo_train | metrics_logged mode=eval step=1150
1185
  2026-04-11 21:51:41,218 | INFO | train_grpo_train | metrics_logged mode=train step=1151
1186
+ 2026-04-11 21:51:45,958 | INFO | train_grpo_train | metrics_logged mode=train step=1152
1187
+ 2026-04-11 21:51:50,666 | INFO | train_grpo_train | metrics_logged mode=train step=1153
1188
+ 2026-04-11 21:51:55,929 | INFO | train_grpo_train | metrics_logged mode=train step=1154
1189
+ 2026-04-11 21:52:02,175 | INFO | train_grpo_train | metrics_logged mode=train step=1155
1190
+ 2026-04-11 21:52:12,180 | INFO | train_grpo_train | metrics_logged mode=train step=1156
1191
+ 2026-04-11 21:52:16,960 | INFO | train_grpo_train | metrics_logged mode=train step=1157
1192
+ 2026-04-11 21:52:21,876 | INFO | train_grpo_train | metrics_logged mode=train step=1158
1193
+ 2026-04-11 21:52:26,984 | INFO | train_grpo_train | metrics_logged mode=train step=1159
1194
+ 2026-04-11 21:52:31,748 | INFO | train_grpo_train | metrics_logged mode=train step=1160
1195
+ 2026-04-11 21:52:36,870 | INFO | train_grpo_train | metrics_logged mode=train step=1161
1196
+ 2026-04-11 21:52:41,664 | INFO | train_grpo_train | metrics_logged mode=train step=1162
1197
+ 2026-04-11 21:52:47,327 | INFO | train_grpo_train | metrics_logged mode=train step=1163
1198
+ 2026-04-11 21:52:52,041 | INFO | train_grpo_train | metrics_logged mode=train step=1164
1199
+ 2026-04-11 21:52:56,812 | INFO | train_grpo_train | metrics_logged mode=train step=1165
1200
+ 2026-04-11 21:53:01,606 | INFO | train_grpo_train | metrics_logged mode=train step=1166
1201
+ 2026-04-11 21:53:06,106 | INFO | train_grpo_train | metrics_logged mode=train step=1167
1202
+ 2026-04-11 21:53:11,234 | INFO | train_grpo_train | metrics_logged mode=train step=1168
1203
+ 2026-04-11 21:53:16,044 | INFO | train_grpo_train | metrics_logged mode=train step=1169
1204
+ 2026-04-11 21:53:26,342 | INFO | train_grpo_train | metrics_logged mode=train step=1170
1205
+ 2026-04-11 21:53:31,041 | INFO | train_grpo_train | metrics_logged mode=train step=1171
1206
+ 2026-04-11 21:53:36,542 | INFO | train_grpo_train | metrics_logged mode=train step=1172
1207
+ 2026-04-11 21:53:41,216 | INFO | train_grpo_train | metrics_logged mode=train step=1173
1208
+ 2026-04-11 21:53:45,682 | INFO | train_grpo_train | metrics_logged mode=train step=1174
1209
+ 2026-04-11 21:53:50,508 | INFO | train_grpo_train | metrics_logged mode=train step=1175
1210
+ 2026-04-11 21:53:58,498 | INFO | train_grpo_train | metrics_logged mode=train step=1176
1211
+ 2026-04-11 21:54:03,433 | INFO | train_grpo_train | metrics_logged mode=train step=1177
1212
+ 2026-04-11 21:54:08,519 | INFO | train_grpo_train | metrics_logged mode=train step=1178
1213
+ 2026-04-11 21:54:14,484 | INFO | train_grpo_train | metrics_logged mode=train step=1179
1214
+ 2026-04-11 21:54:20,576 | INFO | train_grpo_train | metrics_logged mode=train step=1180
1215
+ 2026-04-11 21:54:25,387 | INFO | train_grpo_train | metrics_logged mode=train step=1181
1216
+ 2026-04-11 21:54:31,888 | INFO | train_grpo_train | metrics_logged mode=train step=1182
1217
+ 2026-04-11 21:54:36,937 | INFO | train_grpo_train | metrics_logged mode=train step=1183
1218
+ 2026-04-11 21:54:42,468 | INFO | train_grpo_train | metrics_logged mode=train step=1184
1219
+ 2026-04-11 21:54:46,879 | INFO | train_grpo_train | metrics_logged mode=train step=1185
1220
+ 2026-04-11 21:54:52,265 | INFO | train_grpo_train | metrics_logged mode=train step=1186
1221
+ 2026-04-11 21:54:57,578 | INFO | train_grpo_train | metrics_logged mode=train step=1187
1222
+ 2026-04-11 21:55:02,682 | INFO | train_grpo_train | metrics_logged mode=train step=1188
1223
+ 2026-04-11 21:55:07,556 | INFO | train_grpo_train | metrics_logged mode=train step=1189
1224
+ 2026-04-11 21:55:11,936 | INFO | train_grpo_train | metrics_logged mode=train step=1190
1225
+ 2026-04-11 21:55:17,020 | INFO | train_grpo_train | metrics_logged mode=train step=1191
1226
+ 2026-04-11 21:55:21,903 | INFO | train_grpo_train | metrics_logged mode=train step=1192
1227
+ 2026-04-11 21:55:26,567 | INFO | train_grpo_train | metrics_logged mode=train step=1193
1228
+ 2026-04-11 21:55:32,988 | INFO | train_grpo_train | metrics_logged mode=train step=1194
1229
+ 2026-04-11 21:55:38,909 | INFO | train_grpo_train | metrics_logged mode=train step=1195
1230
+ 2026-04-11 21:55:44,311 | INFO | train_grpo_train | metrics_logged mode=train step=1196
1231
+ 2026-04-11 21:55:48,986 | INFO | train_grpo_train | metrics_logged mode=train step=1197
1232
+ 2026-04-11 21:55:58,748 | INFO | train_grpo_train | metrics_logged mode=train step=1198
1233
+ 2026-04-11 21:56:04,093 | INFO | train_grpo_train | metrics_logged mode=train step=1199
1234
+ 2026-04-11 21:56:08,483 | INFO | train_grpo_train | metrics_logged mode=train step=1200
1235
+ 2026-04-11 21:57:39,205 | INFO | train_grpo_train | metrics_logged mode=eval step=1200
1236
+ 2026-04-11 21:57:46,763 | INFO | train_grpo_train | metrics_logged mode=train step=1201
train_stdout.log CHANGED
The diff for this file is too large to render. See raw diff