Instructions to use Shaer-AI-2/Shaer-adapters-grpo with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use Shaer-AI-2/Shaer-adapters-grpo with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("Shaer-AI-2/Shaer-adapters-grpo", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Training in progress, step 50
Browse files- README.md +1 -1
- adapter_config.json +4 -4
- adapter_model.safetensors +1 -1
- all_generations.jsonl +0 -0
- checkpoint_events.jsonl +2 -5
- config_snapshot.json +1 -1
- dataset_summary.json +54 -26
- effective_runtime_config.json +15 -15
- env_snapshot_masked.json +5 -4
- lineage.json +5 -5
- metrics.csv +54 -6
- metrics.jsonl +53 -5
- plots/chain_metrics.jsonl +53 -5
- plots/chain_runs.json +3 -3
- plots/reward_panels_eval_chain.png +2 -2
- plots/reward_panels_eval_run.png +2 -2
- plots/reward_panels_train_chain.png +2 -2
- plots/reward_panels_train_run.png +2 -2
- plotter.log +144 -52
- plotter.pid +1 -1
- resume_decision.json +3 -3
- reward_exact_count_bonus_debug.jsonl +0 -0
- reward_meter_debug.jsonl +0 -0
- runtime_snapshot.json +5 -5
- train.log +64 -17
- train.pid +1 -1
- train_stdout.log +0 -0
- training_args.bin +1 -1
- watcher.pid +1 -1
README.md
CHANGED
|
@@ -4,8 +4,8 @@ library_name: transformers
|
|
| 4 |
model_name: Shaer-adapters-grpo
|
| 5 |
tags:
|
| 6 |
- generated_from_trainer
|
| 7 |
-
- trl
|
| 8 |
- grpo
|
|
|
|
| 9 |
licence: license
|
| 10 |
---
|
| 11 |
|
|
|
|
| 4 |
model_name: Shaer-adapters-grpo
|
| 5 |
tags:
|
| 6 |
- generated_from_trainer
|
|
|
|
| 7 |
- grpo
|
| 8 |
+
- trl
|
| 9 |
licence: license
|
| 10 |
---
|
| 11 |
|
adapter_config.json
CHANGED
|
@@ -25,13 +25,13 @@
|
|
| 25 |
"rank_pattern": {},
|
| 26 |
"revision": null,
|
| 27 |
"target_modules": [
|
| 28 |
-
"
|
| 29 |
"q_proj",
|
| 30 |
-
"
|
|
|
|
| 31 |
"v_proj",
|
| 32 |
"gate_proj",
|
| 33 |
-
"
|
| 34 |
-
"up_proj"
|
| 35 |
],
|
| 36 |
"target_parameters": null,
|
| 37 |
"task_type": "CAUSAL_LM",
|
|
|
|
| 25 |
"rank_pattern": {},
|
| 26 |
"revision": null,
|
| 27 |
"target_modules": [
|
| 28 |
+
"up_proj",
|
| 29 |
"q_proj",
|
| 30 |
+
"down_proj",
|
| 31 |
+
"o_proj",
|
| 32 |
"v_proj",
|
| 33 |
"gate_proj",
|
| 34 |
+
"k_proj"
|
|
|
|
| 35 |
],
|
| 36 |
"target_parameters": null,
|
| 37 |
"task_type": "CAUSAL_LM",
|
adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 639691872
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:9086bc20e265776833c531b2b21f7ecccba7a49651347fb297871dbdd597a6a0
|
| 3 |
size 639691872
|
all_generations.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint_events.jsonl
CHANGED
|
@@ -1,5 +1,2 @@
|
|
| 1 |
-
{"timestamp_utc": "2026-04-
|
| 2 |
-
{"timestamp_utc": "2026-04-
|
| 3 |
-
{"timestamp_utc": "2026-04-11T12:38:05Z", "event_type": "evaluation_completed", "global_step": 2, "metrics": {"eval_loss": 0.06863964349031448, "eval_runtime": 14.0, "eval_samples_per_second": 0.571, "eval_steps_per_second": 0.143}}
|
| 4 |
-
{"timestamp_utc": "2026-04-11T12:38:07Z", "event_type": "checkpoint_saved", "global_step": 2, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/sanity_check/sanity_20260411_123535/checkpoint-2", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
|
| 5 |
-
{"timestamp_utc": "2026-04-11T12:38:07Z", "event_type": "train_end", "global_step": 2, "best_model_checkpoint": "/root/workspace/Shaer/grpo/outputs/sanity_check/sanity_20260411_123535/checkpoint-2"}
|
|
|
|
| 1 |
+
{"timestamp_utc": "2026-04-11T15:55:14Z", "event_type": "evaluation_completed", "global_step": 50, "metrics": {"eval_loss": 0.018872085958719254, "eval_runtime": 217.5742, "eval_samples_per_second": 0.478, "eval_steps_per_second": 0.06}}
|
| 2 |
+
{"timestamp_utc": "2026-04-11T15:55:16Z", "event_type": "checkpoint_saved", "global_step": 50, "local_checkpoint_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/checkpoint-50", "hub_model_id": "Shaer-AI/Shaer-adapters-grpo", "expected_hub_prefix": "last-checkpoint"}
|
|
|
|
|
|
|
|
|
config_snapshot.json
CHANGED
|
@@ -102,7 +102,7 @@
|
|
| 102 |
"trainer": {
|
| 103 |
"learning_rate": 1e-05,
|
| 104 |
"per_device_train_batch_size": 1,
|
| 105 |
-
"per_device_eval_batch_size":
|
| 106 |
"gradient_accumulation_steps": 8,
|
| 107 |
"max_steps": 1000,
|
| 108 |
"logging_steps": 1,
|
|
|
|
| 102 |
"trainer": {
|
| 103 |
"learning_rate": 1e-05,
|
| 104 |
"per_device_train_batch_size": 1,
|
| 105 |
+
"per_device_eval_batch_size": 2,
|
| 106 |
"gradient_accumulation_steps": 8,
|
| 107 |
"max_steps": 1000,
|
| 108 |
"logging_steps": 1,
|
dataset_summary.json
CHANGED
|
@@ -6,10 +6,10 @@
|
|
| 6 |
"train_split": "train",
|
| 7 |
"eval_split": "eval",
|
| 8 |
"test_split": "test",
|
| 9 |
-
"train_size":
|
| 10 |
-
"eval_size":
|
| 11 |
"test_size": 204,
|
| 12 |
-
"hard_diagnostic_size":
|
| 13 |
"phase1_max_bayts": "20",
|
| 14 |
"allowed_meters": [],
|
| 15 |
"train_manifest_path": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
|
@@ -17,38 +17,66 @@
|
|
| 17 |
"eval_bank_per_meter_per_bucket": 2,
|
| 18 |
"test_bank_per_meter_per_bucket": 4,
|
| 19 |
"train_length_bucket_counts": {
|
| 20 |
-
"
|
| 21 |
-
"4-6":
|
| 22 |
-
"
|
|
|
|
| 23 |
},
|
| 24 |
"eval_length_bucket_counts": {
|
| 25 |
-
"
|
| 26 |
-
"
|
|
|
|
|
|
|
| 27 |
},
|
| 28 |
"hard_diagnostic_length_bucket_counts": {
|
| 29 |
-
"
|
| 30 |
-
"
|
| 31 |
-
"
|
|
|
|
| 32 |
},
|
| 33 |
"train_base_meter_counts": {
|
| 34 |
-
"ال
|
| 35 |
-
"ال
|
| 36 |
-
"ال
|
| 37 |
-
"ال
|
| 38 |
-
"ال
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 39 |
},
|
| 40 |
"eval_base_meter_counts": {
|
| 41 |
-
"ال
|
| 42 |
-
"ال
|
| 43 |
-
"الرجز":
|
| 44 |
-
"ال
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 45 |
},
|
| 46 |
"hard_diagnostic_base_meter_counts": {
|
| 47 |
-
"ال
|
| 48 |
-
"ال
|
| 49 |
-
"ال
|
| 50 |
-
"ال
|
| 51 |
-
"الر
|
| 52 |
-
"ال
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 53 |
}
|
| 54 |
}
|
|
|
|
| 6 |
"train_split": "train",
|
| 7 |
"eval_split": "eval",
|
| 8 |
"test_split": "test",
|
| 9 |
+
"train_size": 25896,
|
| 10 |
+
"eval_size": 104,
|
| 11 |
"test_size": 204,
|
| 12 |
+
"hard_diagnostic_size": 3126,
|
| 13 |
"phase1_max_bayts": "20",
|
| 14 |
"allowed_meters": [],
|
| 15 |
"train_manifest_path": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
|
|
|
| 17 |
"eval_bank_per_meter_per_bucket": 2,
|
| 18 |
"test_bank_per_meter_per_bucket": 4,
|
| 19 |
"train_length_bucket_counts": {
|
| 20 |
+
"1-3": 17433,
|
| 21 |
+
"4-6": 5251,
|
| 22 |
+
"7-10": 1961,
|
| 23 |
+
"11-20": 1251
|
| 24 |
},
|
| 25 |
"eval_length_bucket_counts": {
|
| 26 |
+
"1-3": 26,
|
| 27 |
+
"4-6": 26,
|
| 28 |
+
"7-10": 26,
|
| 29 |
+
"11-20": 26
|
| 30 |
},
|
| 31 |
"hard_diagnostic_length_bucket_counts": {
|
| 32 |
+
"1-3": 683,
|
| 33 |
+
"4-6": 746,
|
| 34 |
+
"7-10": 658,
|
| 35 |
+
"11-20": 1039
|
| 36 |
},
|
| 37 |
"train_base_meter_counts": {
|
| 38 |
+
"البسيط": 3000,
|
| 39 |
+
"الخفيف": 3000,
|
| 40 |
+
"الرجز": 1901,
|
| 41 |
+
"الرمل": 1760,
|
| 42 |
+
"السريع": 2307,
|
| 43 |
+
"الطويل": 3000,
|
| 44 |
+
"الكامل": 3000,
|
| 45 |
+
"المتقارب": 3000,
|
| 46 |
+
"المجتث": 929,
|
| 47 |
+
"المديد": 160,
|
| 48 |
+
"المنسرح": 589,
|
| 49 |
+
"الهزج": 250,
|
| 50 |
+
"الوافر": 3000
|
| 51 |
},
|
| 52 |
"eval_base_meter_counts": {
|
| 53 |
+
"البسيط": 8,
|
| 54 |
+
"الخفيف": 8,
|
| 55 |
+
"الرجز": 8,
|
| 56 |
+
"الرمل": 8,
|
| 57 |
+
"السريع": 8,
|
| 58 |
+
"الطويل": 8,
|
| 59 |
+
"الكامل": 8,
|
| 60 |
+
"المتقارب": 8,
|
| 61 |
+
"المجتث": 8,
|
| 62 |
+
"المديد": 8,
|
| 63 |
+
"المنسرح": 8,
|
| 64 |
+
"الهزج": 8,
|
| 65 |
+
"الوافر": 8
|
| 66 |
},
|
| 67 |
"hard_diagnostic_base_meter_counts": {
|
| 68 |
+
"البسيط": 256,
|
| 69 |
+
"الخفيف": 256,
|
| 70 |
+
"الرجز": 256,
|
| 71 |
+
"الرمل": 256,
|
| 72 |
+
"السريع": 256,
|
| 73 |
+
"الطويل": 54,
|
| 74 |
+
"الكامل": 256,
|
| 75 |
+
"المتقارب": 256,
|
| 76 |
+
"المجتث": 256,
|
| 77 |
+
"المديد": 256,
|
| 78 |
+
"المنسرح": 256,
|
| 79 |
+
"الهزج": 256,
|
| 80 |
+
"الوافر": 256
|
| 81 |
}
|
| 82 |
}
|
effective_runtime_config.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
| 1 |
{
|
| 2 |
-
"mode": "
|
| 3 |
"lineage": {
|
| 4 |
-
"created_at_utc": "2026-04-
|
| 5 |
-
"run_id": "
|
| 6 |
-
"run_dir": "/root/workspace/Shaer/grpo/outputs/
|
| 7 |
-
"chain_id": "
|
| 8 |
-
"root_run_id": "
|
| 9 |
"parent_run_id": "",
|
| 10 |
"parent_run_dir": "",
|
| 11 |
"run_sequence_index": 0
|
|
@@ -16,10 +16,10 @@
|
|
| 16 |
"train_split": "train",
|
| 17 |
"eval_split": "eval",
|
| 18 |
"test_split": "test",
|
| 19 |
-
"train_size":
|
| 20 |
-
"eval_size":
|
| 21 |
"test_size": 204,
|
| 22 |
-
"hard_diagnostic_size":
|
| 23 |
"phase1_max_bayts": "20",
|
| 24 |
"allowed_meters": [],
|
| 25 |
"train_manifest_path": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
|
@@ -50,19 +50,19 @@
|
|
| 50 |
"max_completion_length": 640,
|
| 51 |
"temperature": 1.0,
|
| 52 |
"top_p": 1.0,
|
| 53 |
-
"num_generations":
|
| 54 |
"num_generations_eval": 2
|
| 55 |
},
|
| 56 |
"effective_trainer": {
|
| 57 |
"learning_rate": 1e-05,
|
| 58 |
"per_device_train_batch_size": 1,
|
| 59 |
-
"per_device_eval_batch_size_configured":
|
| 60 |
-
"per_device_eval_batch_size":
|
| 61 |
"gradient_accumulation_steps": 8,
|
| 62 |
-
"max_steps":
|
| 63 |
"logging_steps": 1,
|
| 64 |
-
"eval_steps":
|
| 65 |
-
"save_steps":
|
| 66 |
"save_total_limit": 4,
|
| 67 |
"beta": 0.02,
|
| 68 |
"bf16": true,
|
|
|
|
| 1 |
{
|
| 2 |
+
"mode": "train",
|
| 3 |
"lineage": {
|
| 4 |
+
"created_at_utc": "2026-04-11T15:46:16Z",
|
| 5 |
+
"run_id": "shaer_grpo_20260411_154531",
|
| 6 |
+
"run_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531",
|
| 7 |
+
"chain_id": "shaer_grpo_20260411_154531",
|
| 8 |
+
"root_run_id": "shaer_grpo_20260411_154531",
|
| 9 |
"parent_run_id": "",
|
| 10 |
"parent_run_dir": "",
|
| 11 |
"run_sequence_index": 0
|
|
|
|
| 16 |
"train_split": "train",
|
| 17 |
"eval_split": "eval",
|
| 18 |
"test_split": "test",
|
| 19 |
+
"train_size": 25896,
|
| 20 |
+
"eval_size": 104,
|
| 21 |
"test_size": 204,
|
| 22 |
+
"hard_diagnostic_size": 3126,
|
| 23 |
"phase1_max_bayts": "20",
|
| 24 |
"allowed_meters": [],
|
| 25 |
"train_manifest_path": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
|
|
|
| 50 |
"max_completion_length": 640,
|
| 51 |
"temperature": 1.0,
|
| 52 |
"top_p": 1.0,
|
| 53 |
+
"num_generations": 8,
|
| 54 |
"num_generations_eval": 2
|
| 55 |
},
|
| 56 |
"effective_trainer": {
|
| 57 |
"learning_rate": 1e-05,
|
| 58 |
"per_device_train_batch_size": 1,
|
| 59 |
+
"per_device_eval_batch_size_configured": 2,
|
| 60 |
+
"per_device_eval_batch_size": 8,
|
| 61 |
"gradient_accumulation_steps": 8,
|
| 62 |
+
"max_steps": 1000,
|
| 63 |
"logging_steps": 1,
|
| 64 |
+
"eval_steps": 50,
|
| 65 |
+
"save_steps": 50,
|
| 66 |
"save_total_limit": 4,
|
| 67 |
"beta": 0.02,
|
| 68 |
"bf16": true,
|
env_snapshot_masked.json
CHANGED
|
@@ -3,7 +3,7 @@
|
|
| 3 |
"SHELL": "/bin/bash",
|
| 4 |
"GRPO_TRAIN_DATASET_ID": "Shaer-AI/ashaar-enhanced-desc-baseform-final-sft-lte20-min500-splits-grpo-meter-count-v1",
|
| 5 |
"COLORTERM": "",
|
| 6 |
-
"GRPO_RUN_DIR": "/root/workspace/Shaer/grpo/outputs/
|
| 7 |
"ELECTRON_RUN_AS_NODE": "1",
|
| 8 |
"GH_PAGER": "cat",
|
| 9 |
"PWD": "/root/workspace/Shaer/grpo",
|
|
@@ -18,7 +18,7 @@
|
|
| 18 |
"VSCODE_AGENT_FOLDER": "/root/.vscode-server",
|
| 19 |
"GRPO_CURATED_MANIFEST_PATH": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
| 20 |
"SSH_CONNECTION": "212.98.144.20 53044 172.18.0.2 22",
|
| 21 |
-
"WATCHER_STEP_EMAIL_EVERY": "
|
| 22 |
"RUNPOD_MEM_GB": "125",
|
| 23 |
"RUNPOD_PUBLIC_IP": "91.199.227.82",
|
| 24 |
"RUNPOD_GPU_COUNT": "1",
|
|
@@ -30,6 +30,7 @@
|
|
| 30 |
"USER": "root",
|
| 31 |
"GIT_PAGER": "cat",
|
| 32 |
"RUNPOD_DC_ID": "EU-NL-1",
|
|
|
|
| 33 |
"SHLVL": "0",
|
| 34 |
"PAGER": "cat",
|
| 35 |
"VSCODE_CWD": "/root",
|
|
@@ -45,9 +46,11 @@
|
|
| 45 |
"CODEX_INTERNAL_ORIGINATOR_OVERRIDE": "codex_vscode",
|
| 46 |
"LC_ALL": "C.UTF-8",
|
| 47 |
"SFT_ADAPTER_REPO": "Shaer-AI/Shaer-adapters",
|
|
|
|
| 48 |
"RUNPOD_API_KEY": "***MASKED***",
|
| 49 |
"BROWSER": "/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/bin/helpers/browser.sh",
|
| 50 |
"PATH": "/root/workspace/Shaer/grpo/.venv/bin:/root/.codex/tmp/arg0/codex-arg0uP3dEp:/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/bin/remote-cli:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin:/root/.vscode-server/extensions/openai.chatgpt-26.409.20454-linux-x64/bin/linux-x86_64",
|
|
|
|
| 51 |
"APPLICATION_INSIGHTS_NO_STATSBEAT": "true",
|
| 52 |
"RUST_LOG": "warn",
|
| 53 |
"VSCODE_NLS_CONFIG": "{\"userLocale\":\"en\",\"osLocale\":\"en\",\"resolvedLanguage\":\"en\",\"defaultMessagesFile\":\"/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/out/nls.messages.json\",\"locale\":\"en\",\"availableLanguages\":{}}",
|
|
@@ -88,8 +91,6 @@
|
|
| 88 |
"REWARD_CACHE_DIR": "./cache/reward_cache",
|
| 89 |
"MEANING_FIT_PROMPT_FILE": "prompts/meaning_fit.yaml",
|
| 90 |
"MEANING_SUBSTANCE_PROMPT_FILE": "prompts/meaning_substance.yaml",
|
| 91 |
-
"GRPO_RESUME_MODE": "auto",
|
| 92 |
-
"GRPO_RESUME_PATH": "",
|
| 93 |
"WATCHER_STATE_DIR": "./watcher_state",
|
| 94 |
"CUDA_MODULE_LOADING": "LAZY",
|
| 95 |
"TORCHINDUCTOR_CACHE_DIR": "/tmp/torchinductor_root"
|
|
|
|
| 3 |
"SHELL": "/bin/bash",
|
| 4 |
"GRPO_TRAIN_DATASET_ID": "Shaer-AI/ashaar-enhanced-desc-baseform-final-sft-lte20-min500-splits-grpo-meter-count-v1",
|
| 5 |
"COLORTERM": "",
|
| 6 |
+
"GRPO_RUN_DIR": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531",
|
| 7 |
"ELECTRON_RUN_AS_NODE": "1",
|
| 8 |
"GH_PAGER": "cat",
|
| 9 |
"PWD": "/root/workspace/Shaer/grpo",
|
|
|
|
| 18 |
"VSCODE_AGENT_FOLDER": "/root/.vscode-server",
|
| 19 |
"GRPO_CURATED_MANIFEST_PATH": "/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv",
|
| 20 |
"SSH_CONNECTION": "212.98.144.20 53044 172.18.0.2 22",
|
| 21 |
+
"WATCHER_STEP_EMAIL_EVERY": "50",
|
| 22 |
"RUNPOD_MEM_GB": "125",
|
| 23 |
"RUNPOD_PUBLIC_IP": "91.199.227.82",
|
| 24 |
"RUNPOD_GPU_COUNT": "1",
|
|
|
|
| 30 |
"USER": "root",
|
| 31 |
"GIT_PAGER": "cat",
|
| 32 |
"RUNPOD_DC_ID": "EU-NL-1",
|
| 33 |
+
"GRPO_RESUME_MODE": "fresh",
|
| 34 |
"SHLVL": "0",
|
| 35 |
"PAGER": "cat",
|
| 36 |
"VSCODE_CWD": "/root",
|
|
|
|
| 46 |
"CODEX_INTERNAL_ORIGINATOR_OVERRIDE": "codex_vscode",
|
| 47 |
"LC_ALL": "C.UTF-8",
|
| 48 |
"SFT_ADAPTER_REPO": "Shaer-AI/Shaer-adapters",
|
| 49 |
+
"CODEX_SANDBOX_NETWORK_DISABLED": "1",
|
| 50 |
"RUNPOD_API_KEY": "***MASKED***",
|
| 51 |
"BROWSER": "/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/bin/helpers/browser.sh",
|
| 52 |
"PATH": "/root/workspace/Shaer/grpo/.venv/bin:/root/.codex/tmp/arg0/codex-arg0uP3dEp:/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/bin/remote-cli:/usr/local/nvidia/bin:/usr/local/cuda/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin:/root/.vscode-server/extensions/openai.chatgpt-26.409.20454-linux-x64/bin/linux-x86_64",
|
| 53 |
+
"GRPO_RESUME_PATH": "",
|
| 54 |
"APPLICATION_INSIGHTS_NO_STATSBEAT": "true",
|
| 55 |
"RUST_LOG": "warn",
|
| 56 |
"VSCODE_NLS_CONFIG": "{\"userLocale\":\"en\",\"osLocale\":\"en\",\"resolvedLanguage\":\"en\",\"defaultMessagesFile\":\"/root/.vscode-server/cli/servers/Stable-41dd792b5e652393e7787322889ed5fdc58bd75b/server/out/nls.messages.json\",\"locale\":\"en\",\"availableLanguages\":{}}",
|
|
|
|
| 91 |
"REWARD_CACHE_DIR": "./cache/reward_cache",
|
| 92 |
"MEANING_FIT_PROMPT_FILE": "prompts/meaning_fit.yaml",
|
| 93 |
"MEANING_SUBSTANCE_PROMPT_FILE": "prompts/meaning_substance.yaml",
|
|
|
|
|
|
|
| 94 |
"WATCHER_STATE_DIR": "./watcher_state",
|
| 95 |
"CUDA_MODULE_LOADING": "LAZY",
|
| 96 |
"TORCHINDUCTOR_CACHE_DIR": "/tmp/torchinductor_root"
|
lineage.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
-
"created_at_utc": "2026-04-
|
| 3 |
-
"run_id": "
|
| 4 |
-
"run_dir": "/root/workspace/Shaer/grpo/outputs/
|
| 5 |
-
"chain_id": "
|
| 6 |
-
"root_run_id": "
|
| 7 |
"parent_run_id": "",
|
| 8 |
"parent_run_dir": "",
|
| 9 |
"run_sequence_index": 0
|
|
|
|
| 1 |
{
|
| 2 |
+
"created_at_utc": "2026-04-11T15:46:16Z",
|
| 3 |
+
"run_id": "shaer_grpo_20260411_154531",
|
| 4 |
+
"run_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531",
|
| 5 |
+
"chain_id": "shaer_grpo_20260411_154531",
|
| 6 |
+
"root_run_id": "shaer_grpo_20260411_154531",
|
| 7 |
"parent_run_id": "",
|
| 8 |
"parent_run_dir": "",
|
| 9 |
"run_sequence_index": 0
|
metrics.csv
CHANGED
|
@@ -1,6 +1,54 @@
|
|
| 1 |
-
clip_ratio/high_max,clip_ratio/high_mean,clip_ratio/low_mean,clip_ratio/low_min,clip_ratio/region_mean,completions/clipped_ratio,completions/max_length,completions/max_terminated_length,completions/mean_length,completions/mean_terminated_length,completions/min_length,completions/min_terminated_length,entropy,epoch,eval_clip_ratio/high_max,eval_clip_ratio/high_mean,eval_clip_ratio/low_mean,eval_clip_ratio/low_min,eval_clip_ratio/region_mean,eval_completions/clipped_ratio,eval_completions/max_length,eval_completions/max_terminated_length,eval_completions/mean_length,eval_completions/mean_terminated_length,eval_completions/min_length,eval_completions/min_terminated_length,eval_entropy,eval_frac_reward_zero_std,eval_kl,eval_loss,eval_num_tokens,eval_reward,eval_reward_exact_count_bonus_mean,eval_reward_exact_count_bonus_std,eval_reward_meter_mean,eval_reward_meter_std,eval_reward_std,eval_reward_total_mean,eval_rewards/exact_count_bonus/mean,eval_rewards/exact_count_bonus/std,eval_rewards/meter/mean,eval_rewards/meter/std,eval_runtime,eval_samples_per_second,eval_sampling/importance_sampling_ratio/max,eval_sampling/importance_sampling_ratio/mean,eval_sampling/importance_sampling_ratio/min,eval_sampling/sampling_logp_difference/max,eval_sampling/sampling_logp_difference/mean,eval_steps_per_second,frac_reward_zero_std,global_step,grad_norm,kl,learning_rate,loss,mode,num_tokens,reward,reward_exact_count_bonus_mean,reward_exact_count_bonus_std,reward_meter_mean,reward_meter_std,reward_std,reward_total_mean,rewards/exact_count_bonus/mean,rewards/exact_count_bonus/std,rewards/meter/mean,rewards/meter/std,sampling/importance_sampling_ratio/max,sampling/importance_sampling_ratio/mean,sampling/importance_sampling_ratio/min,sampling/sampling_logp_difference/max,sampling/sampling_logp_difference/mean,timestamp_utc
|
| 2 |
-
0.
|
| 3 |
-
|
| 4 |
-
0.
|
| 5 |
-
|
| 6 |
-
,,,,,,,,,,,,
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
clip_ratio/high_max,clip_ratio/high_mean,clip_ratio/low_mean,clip_ratio/low_min,clip_ratio/region_mean,completions/clipped_ratio,completions/max_length,completions/max_terminated_length,completions/mean_length,completions/mean_terminated_length,completions/min_length,completions/min_terminated_length,entropy,epoch,eval_clip_ratio/high_max,eval_clip_ratio/high_mean,eval_clip_ratio/low_mean,eval_clip_ratio/low_min,eval_clip_ratio/region_mean,eval_completions/clipped_ratio,eval_completions/max_length,eval_completions/max_terminated_length,eval_completions/mean_length,eval_completions/mean_terminated_length,eval_completions/min_length,eval_completions/min_terminated_length,eval_entropy,eval_frac_reward_zero_std,eval_kl,eval_loss,eval_num_tokens,eval_reward,eval_reward_exact_count_bonus_mean,eval_reward_exact_count_bonus_std,eval_reward_meter_mean,eval_reward_meter_std,eval_reward_std,eval_reward_total_mean,eval_rewards/exact_count_bonus/mean,eval_rewards/exact_count_bonus/std,eval_rewards/meter/mean,eval_rewards/meter/std,eval_runtime,eval_samples_per_second,eval_sampling/importance_sampling_ratio/max,eval_sampling/importance_sampling_ratio/mean,eval_sampling/importance_sampling_ratio/min,eval_sampling/sampling_logp_difference/max,eval_sampling/sampling_logp_difference/mean,eval_steps_per_second,frac_reward_zero_std,global_step,grad_norm,kl,learning_rate,loss,mode,num_tokens,reward,reward_exact_count_bonus_mean,reward_exact_count_bonus_std,reward_meter_mean,reward_meter_std,reward_std,reward_total_mean,rewards/exact_count_bonus/mean,rewards/exact_count_bonus/std,rewards/meter/mean,rewards/meter/std,sampling/importance_sampling_ratio/max,sampling/importance_sampling_ratio/mean,sampling/importance_sampling_ratio/min,sampling/sampling_logp_difference/max,sampling/sampling_logp_difference/mean,timestamp_utc
|
| 2 |
+
0.16360880248248577,0.16360880248248577,0.09221794456243515,0.09221794456243515,0.2558267470449209,0.0,83.0,83.0,72.125,72.125,68.0,68.0,2.8896071165800095,3.861600247142416e-05,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,1,14.433236122131348,1.716715857386589,1e-05,0.0875,train,2073.0,0.9467406272888184,1.0,0.0,0.7467405796051025,0.2669501304626465,0.26695016026496887,0.9467406272888184,1.0,0.0,0.7467405796051025,0.2669501304626465,2.0,0.9819375872612,0.09113290905952454,2.3954362869262695,0.1810092180967331,2026-04-11T15:46:43Z
|
| 3 |
+
0.15457439702004194,0.15457439702004194,0.037109375,0.037109375,0.19168377202004194,0.0,67.0,67.0,59.25,59.25,44.0,44.0,1.2240338400006294,7.723200494284832e-05,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,2,14.803139686584473,1.5766338407993317,9.990000000000001e-06,0.1013,train,3899.0,0.8921604156494141,1.0,0.0,0.692160427570343,0.3035436272621155,0.3035435974597931,0.8921604156494141,1.0,0.0,0.692160427570343,0.3035436272621155,2.0,0.9748801589012146,0.27146539092063904,1.6599078178405762,0.17282956838607788,2026-04-11T15:46:49Z
|
| 4 |
+
0.11026187054812908,0.11026187054812908,0.08594596944749355,0.08594596944749355,0.19620783999562263,0.0,51.0,51.0,40.625,40.625,32.0,32.0,2.205263152718544,0.00011584800741427248,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,3,22.877466201782227,2.4038615822792053,9.980000000000001e-06,0.1019,train,5552.0,0.7208516597747803,1.0,0.0,0.5208516120910645,0.4123581349849701,0.4123581051826477,0.7208516597747803,1.0,0.0,0.5208516120910645,0.4123581349849701,1.952972412109375,0.9614238142967224,0.26364248991012573,1.3331613540649414,0.18985480070114136,2026-04-11T15:46:53Z
|
| 5 |
+
0.09267808869481087,0.09267808869481087,0.04130434803664684,0.04130434803664684,0.1339824367314577,0.0,26.0,26.0,22.25,22.25,19.0,19.0,1.1889886930584908,0.00015446400988569664,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,4,23.66639518737793,1.5658123940229416,9.970000000000001e-06,0.0914,train,6962.0,0.9492155313491821,1.0,0.0,0.7492154836654663,0.4238397777080536,0.4238397479057312,0.9492155313491821,1.0,0.0,0.7492154836654663,0.4238397777080536,2.0,1.0157339572906494,0.507404625415802,1.2559912204742432,0.11156373471021652,2026-04-11T15:46:58Z
|
| 6 |
+
0.07602224312722683,0.07602224312722683,0.14424454979598522,0.14424454979598522,0.22026679292321205,0.0,235.0,235.0,152.375,152.375,99.0,99.0,3.3276279270648956,0.0001930800123571208,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,5,9.979012489318848,1.233508050441742,9.960000000000001e-06,-0.0187,train,9821.0,0.5120292901992798,1.0,0.0,0.31202924251556396,0.2894730269908905,0.2894730269908905,0.5120292901992798,1.0,0.0,0.31202924251556396,0.2894730269908905,2.0,0.987079381942749,0.340789794921875,1.489715576171875,0.19261623919010162,2026-04-11T15:47:07Z
|
| 7 |
+
0.1323312446475029,0.1323312446475029,0.09787485934793949,0.09787485934793949,0.2302061039954424,0.0,100.0,100.0,76.5,76.5,61.0,61.0,3.393460273742676,0.00023169601482854495,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,6,15.483458518981934,2.003227472305298,9.950000000000001e-06,0.1184,train,11697.0,0.9063730835914612,1.0,0.0,0.7063730955123901,0.21755146980285645,0.21755146980285645,0.9063730835914612,1.0,0.0,0.7063730955123901,0.21755146980285645,2.0,0.9750673770904541,0.2625133991241455,1.3374531269073486,0.21761518716812134,2026-04-11T15:47:12Z
|
| 8 |
+
0.07538295164704323,0.07538295164704323,0.12269405275583267,0.12269405275583267,0.1980770044028759,0.0,135.0,135.0,93.875,93.875,62.0,62.0,2.9779615700244904,0.0002703120172999691,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,7,16.140308380126953,2.0300692468881607,9.940000000000001e-06,0.2978,train,13816.0,0.5870001316070557,0.875,0.3535533845424652,0.4120001196861267,0.33185702562332153,0.36883509159088135,0.5870001316070557,0.875,0.3535533845424652,0.4120001196861267,0.33185702562332153,2.0,0.9866839051246643,0.18673592805862427,1.7315394878387451,0.21371272206306458,2026-04-11T15:47:19Z
|
| 9 |
+
0.10601851902902126,0.10601851902902126,0.09676836617290974,0.09676836617290974,0.202786885201931,0.0,54.0,54.0,42.0,42.0,28.0,28.0,2.9420621395111084,0.00030892801977139327,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,8,20.1074161529541,2.0594891011714935,9.930000000000001e-06,0.0345,train,15480.0,0.7437530755996704,0.875,0.3535533845424652,0.5687531232833862,0.3685603737831116,0.39819100499153137,0.7437530755996704,0.875,0.3535533845424652,0.5687531232833862,0.3685603737831116,2.0,0.9912540316581726,0.1432340294122696,1.9432754516601562,0.19006481766700745,2026-04-11T15:47:24Z
|
| 10 |
+
0.14036413840949535,0.14036413840949535,0.0800175815820694,0.0800175815820694,0.22038171999156475,0.0,67.0,67.0,47.25,47.25,37.0,37.0,2.3184936344623566,0.00034754402224281743,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,9,21.095970153808594,1.5875789374113083,9.920000000000002e-06,0.1313,train,17114.0,0.7934404015541077,1.0,0.0,0.5934403538703918,0.4387426972389221,0.4387427270412445,0.7934404015541077,1.0,0.0,0.5934403538703918,0.4387426972389221,2.0,0.9886912107467651,0.10704635828733444,2.2344932556152344,0.18523243069648743,2026-04-11T15:47:29Z
|
| 11 |
+
0.09042712114751339,0.09042712114751339,0.16380306333303452,0.16380306333303452,0.2542301844805479,0.0,62.0,62.0,48.125,48.125,33.0,33.0,3.0384787023067474,0.0003861600247142416,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,10,17.000186920166016,1.8380893766880035,9.91e-06,-0.0138,train,18851.0,0.6637843251228333,0.875,0.3535533845424652,0.4887843132019043,0.42971470952033997,0.45756226778030396,0.6637843251228333,0.875,0.3535533845424652,0.4887843132019043,0.42971470952033997,2.0,0.9984660148620605,0.3167512118816376,1.3021516799926758,0.2051706463098526,2026-04-11T15:47:34Z
|
| 12 |
+
0.08737440872937441,0.08737440872937441,0.11223171092569828,0.11223171092569828,0.1996061196550727,0.0,59.0,59.0,44.875,44.875,29.0,29.0,2.5153299272060394,0.00042477602718566575,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,11,16.804243087768555,1.717022493481636,9.9e-06,0.1831,train,20578.0,0.506911039352417,1.0,0.0,0.30691099166870117,0.4142797291278839,0.4142796993255615,0.506911039352417,1.0,0.0,0.30691099166870117,0.4142797291278839,2.0,0.9988991618156433,0.14744210243225098,1.9143197536468506,0.18619823455810547,2026-04-11T15:47:39Z
|
| 13 |
+
0.048309178091585636,0.048309178091585636,0.13612079434096813,0.13612079434096813,0.18442997243255377,0.0,72.0,72.0,58.0,58.0,38.0,38.0,1.546494573354721,0.0004633920296570899,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,12,22.950233459472656,2.302564397454262,9.89e-06,-0.02,train,22530.0,0.4489539563655853,1.0,0.0,0.2489539384841919,0.23509462177753448,0.23509462177753448,0.4489539563655853,1.0,0.0,0.2489539384841919,0.23509462177753448,2.0,0.9429361820220947,0.07527641206979752,3.918379306793213,0.2930803596973419,2026-04-11T15:47:44Z
|
| 14 |
+
0.11046412773430347,0.11046412773430347,0.0789086427539587,0.0789086427539587,0.18937277048826218,0.0,61.0,61.0,46.875,46.875,33.0,33.0,2.7278781831264496,0.000502008032128514,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,13,20.975563049316406,2.4146449714899063,9.88e-06,0.0551,train,24249.0,0.7702009677886963,1.0,0.0,0.5702009201049805,0.416922926902771,0.41692298650741577,0.7702009677886963,1.0,0.0,0.5702009201049805,0.416922926902771,2.0,0.9821399450302124,0.20114654302597046,1.6037216186523438,0.19165004789829254,2026-04-11T15:47:49Z
|
| 15 |
+
0.09134172648191452,0.09134172648191452,0.13307647220790386,0.13307647220790386,0.22441819868981838,0.0,126.0,126.0,82.5,82.5,63.0,63.0,2.838075205683708,0.0005406240345999382,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,14,13.641975402832031,1.3619689494371414,9.87e-06,0.0636,train,26453.0,0.5489101409912109,1.0,0.0,0.3489101529121399,0.34769579768180847,0.34769579768180847,0.5489101409912109,1.0,0.0,0.3489101529121399,0.34769579768180847,2.0,0.9998252987861633,0.1022249385714531,2.2805795669555664,0.2145778238773346,2026-04-11T15:47:56Z
|
| 16 |
+
0.11765297874808311,0.11765297874808311,0.05330882500857115,0.05330882500857115,0.17096180375665426,0.0,31.0,31.0,22.75,22.75,15.0,15.0,2.376413881778717,0.0005792400370713623,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,15,26.442886352539062,1.7645654305815697,9.86e-06,0.1104,train,27811.0,0.8123704195022583,1.0,0.0,0.6123703718185425,0.42484015226364136,0.42484015226364136,0.8123704195022583,1.0,0.0,0.6123703718185425,0.42484015226364136,2.0,0.9939978122711182,0.3356194794178009,2.0358657836914062,0.21215009689331055,2026-04-11T15:48:00Z
|
| 17 |
+
0.10188646428287029,0.10188646428287029,0.10445018857717514,0.10445018857717514,0.20633665286004543,0.0,41.0,41.0,34.875,34.875,29.0,29.0,2.617787539958954,0.0006178560395427865,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,16,20.25496482849121,1.90836800634861,9.85e-06,0.105,train,29402.0,0.7948991060256958,1.0,0.0,0.59489905834198,0.4257949888706207,0.42579495906829834,0.7948991060256958,1.0,0.0,0.59489905834198,0.4257949888706207,2.0,0.967242956161499,0.24731610715389252,1.4742670059204102,0.2184409201145172,2026-04-11T15:48:05Z
|
| 18 |
+
0.056653511710464954,0.056653511710464954,0.059752749279141426,0.059752749279141426,0.11640626098960638,0.0,49.0,49.0,41.125,41.125,26.0,26.0,0.8714146893471479,0.0006564720420142106,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,17,25.74098014831543,1.7420838475227356,9.84e-06,0.1285,train,30963.0,1.050354242324829,1.0,0.0,0.8503542542457581,0.3227464556694031,0.32274648547172546,1.050354242324829,1.0,0.0,0.8503542542457581,0.3227464556694031,2.0,0.9962012767791748,0.22342658042907715,1.4986724853515625,0.10933797806501389,2026-04-11T15:48:09Z
|
| 19 |
+
0.07529962994158268,0.07529962994158268,0.11914125084877014,0.11914125084877014,0.19444088079035282,0.0,61.0,61.0,42.625,42.625,31.0,31.0,2.5996862947940826,0.0006950880444856349,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,18,17.443212509155273,1.4733051806688309,9.83e-06,0.1207,train,32520.0,0.5162518620491028,1.0,0.0,0.31625181436538696,0.2878619432449341,0.2878619432449341,0.5162518620491028,1.0,0.0,0.31625181436538696,0.2878619432449341,2.0,0.9975839257240295,0.3923143446445465,0.9356918334960938,0.1679675132036209,2026-04-11T15:48:14Z
|
| 20 |
+
0.08858394995331764,0.08858394995331764,0.11502809636294842,0.11502809636294842,0.20361204631626606,0.0,169.0,169.0,119.0,119.0,67.0,67.0,2.8221229016780853,0.000733704046957059,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,19,8.90630054473877,1.0056203603744507,9.820000000000001e-06,0.1359,train,34880.0,0.6386695504188538,0.75,0.4629100561141968,0.4886695444583893,0.21834495663642883,0.25513508915901184,0.6386695504188538,0.75,0.4629100561141968,0.4886695444583893,0.21834495663642883,2.0,1.001911997795105,0.27288922667503357,1.298689365386963,0.16941869258880615,2026-04-11T15:48:22Z
|
| 21 |
+
0.12786824442446232,0.12786824442446232,0.1128854975104332,0.1128854975104332,0.24075374193489552,0.0,73.0,73.0,53.125,53.125,35.0,35.0,2.9686510264873505,0.0007723200494284832,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,20,14.272367477416992,1.500285528600216,9.810000000000001e-06,0.1452,train,36697.0,0.7846182584762573,1.0,0.0,0.584618330001831,0.37843555212020874,0.37843549251556396,0.7846182584762573,1.0,0.0,0.584618330001831,0.37843555212020874,2.0,0.9827072620391846,0.1804894655942917,1.712082862854004,0.18555279076099396,2026-04-11T15:48:27Z
|
| 22 |
+
0.08470636792480946,0.08470636792480946,0.1339168418198824,0.1339168418198824,0.21862320974469185,0.0,163.0,163.0,127.875,127.875,88.0,88.0,3.330483376979828,0.0008109360518999073,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,21,8.9111328125,1.3634046390652657,9.800000000000001e-06,0.0602,train,39216.0,0.7299741506576538,1.0,0.0,0.5299741625785828,0.38676267862319946,0.38676267862319946,0.7299741506576538,1.0,0.0,0.5299741625785828,0.38676267862319946,2.0,0.9976959228515625,0.23953956365585327,1.4290367364883423,0.1731364130973816,2026-04-11T15:48:34Z
|
| 23 |
+
0.11499525140970945,0.11499525140970945,0.09037221781909466,0.09037221781909466,0.2053674692288041,0.0,50.0,50.0,41.875,41.875,32.0,32.0,1.974789410829544,0.0008495520543713315,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,22,19.450031280517578,1.744483582675457,9.790000000000001e-06,0.0956,train,40839.0,0.9068014621734619,1.0,0.0,0.7068014740943909,0.39332520961761475,0.39332520961761475,0.9068014621734619,1.0,0.0,0.7068014740943909,0.39332520961761475,2.0,0.9838151931762695,0.2812429666519165,1.2685363292694092,0.19435128569602966,2026-04-11T15:48:39Z
|
| 24 |
+
0.0615717563778162,0.0615717563778162,0.12430914491415024,0.12430914491415024,0.18588090129196644,0.0,72.0,72.0,59.5,59.5,51.0,51.0,2.614344671368599,0.0008881680568427556,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,23,11.986440658569336,1.2204247117042542,9.780000000000001e-06,-0.0134,train,42619.0,0.5556836128234863,1.0,0.0,0.3556836247444153,0.41377386450767517,0.4137738347053528,0.5556836128234863,1.0,0.0,0.3556836247444153,0.41377386450767517,2.0,0.9997124671936035,0.37629470229148865,0.9905836582183838,0.15584221482276917,2026-04-11T15:48:44Z
|
| 25 |
+
0.08839947171509266,0.08839947171509266,0.10080079734325409,0.10080079734325409,0.18920026905834675,0.0,68.0,68.0,54.75,54.75,29.0,29.0,2.8385555744171143,0.0009267840593141798,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,24,15.669774055480957,1.9536623805761337,9.770000000000001e-06,0.0328,train,44409.0,0.6911630034446716,0.875,0.3535533845424652,0.5161629915237427,0.41966450214385986,0.40840157866477966,0.6911630034446716,0.875,0.3535533845424652,0.5161629915237427,0.41966450214385986,2.0,1.0036118030548096,0.2979213297367096,1.210925817489624,0.1744653582572937,2026-04-11T15:48:49Z
|
| 26 |
+
0.035714286379516125,0.035714286379516125,0.12277928367257118,0.12277928367257118,0.1584935700520873,0.0,56.0,56.0,36.125,36.125,25.0,25.0,1.6488263756036758,0.0009654000617856039,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,25,13.490758895874023,1.5680749863386154,9.760000000000001e-06,0.056,train,46042.0,0.4736630320549011,1.0,0.0,0.2736630439758301,0.30189424753189087,0.3018941879272461,0.4736630320549011,1.0,0.0,0.2736630439758301,0.30189424753189087,2.0,0.9891919493675232,0.3343762159347534,1.1764202117919922,0.13724787533283234,2026-04-11T15:48:54Z
|
| 27 |
+
0.07951899617910385,0.07951899617910385,0.10312438476830721,0.10312438476830721,0.18264338094741106,0.0,67.0,67.0,44.875,44.875,28.0,28.0,2.3349276185035706,0.001004016064257028,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,26,16.0242919921875,1.6573083698749542,9.75e-06,0.202,train,47681.0,0.6207371950149536,1.0,0.0,0.42073720693588257,0.3384658396244049,0.3384658098220825,0.6207371950149536,1.0,0.0,0.42073720693588257,0.3384658396244049,2.0,0.9966971278190613,0.2744143009185791,1.2931162118911743,0.17049100995063782,2026-04-11T15:49:00Z
|
| 28 |
+
0.11580519005656242,0.11580519005656242,0.11959273181855679,0.11959273181855679,0.2353979218751192,0.0,157.0,157.0,114.625,114.625,79.0,79.0,2.8878106623888016,0.0010426320667284523,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,27,21.260799407958984,2.746354952454567,9.74e-06,0.1939,train,50118.0,0.5957071781158447,0.875,0.3535533845424652,0.42070716619491577,0.32767125964164734,0.3487240672111511,0.5957071781158447,0.875,0.3535533845424652,0.42070716619491577,0.32767125964164734,2.0,0.9976061582565308,0.15407966077327728,1.8702855110168457,0.19498828053474426,2026-04-11T15:49:07Z
|
| 29 |
+
0.0886442456394434,0.0886442456394434,0.12361728027462959,0.12361728027462959,0.212261525914073,0.0,117.0,117.0,102.0,102.0,86.0,86.0,1.6325490325689316,0.0010812480691998764,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,28,25.492298126220703,2.430370479822159,9.73e-06,0.1229,train,52550.0,0.6697388291358948,0.875,0.3535533845424652,0.4947388172149658,0.2714572250843048,0.23900270462036133,0.6697388291358948,0.875,0.3535533845424652,0.4947388172149658,0.2714572250843048,2.0,0.9748236536979675,0.062137819826602936,2.778400421142578,0.2764182388782501,2026-04-11T15:49:13Z
|
| 30 |
+
0.15308464504778385,0.15308464504778385,0.07247674837708473,0.07247674837708473,0.22556139342486858,0.0,71.0,71.0,43.25,43.25,34.0,34.0,2.0946053713560104,0.0011198640716713006,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,29,23.189964294433594,1.8857728093862534,9.72e-06,-0.0165,train,54344.0,0.8811927437782288,1.0,0.0,0.6811927556991577,0.32601746916770935,0.3260175287723541,0.8811927437782288,1.0,0.0,0.6811927556991577,0.32601746916770935,2.0,1.0205129384994507,0.2808680534362793,1.2698702812194824,0.18413126468658447,2026-04-11T15:49:18Z
|
| 31 |
+
0.1298079490661621,0.1298079490661621,0.093433802947402,0.093433802947402,0.2232417520135641,0.0,234.0,234.0,146.75,146.75,71.0,71.0,3.4500816762447357,0.0011584800741427247,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,30,9.213912963867188,1.0472772866487503,9.71e-06,-0.1499,train,57126.0,0.5599656701087952,1.0,0.0,0.3599656820297241,0.22725822031497955,0.22725820541381836,0.5599656701087952,1.0,0.0,0.3599656820297241,0.22725822031497955,2.0,1.0062932968139648,0.24727565050125122,1.3972516059875488,0.17979256808757782,2026-04-11T15:49:27Z
|
| 32 |
+
0.09737317077815533,0.09737317077815533,0.12893840484321117,0.12893840484321117,0.2263115756213665,0.0,64.0,64.0,56.625,56.625,49.0,49.0,2.848589450120926,0.001197096076614149,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,31,20.018091201782227,1.5809837579727173,9.7e-06,0.0747,train,58875.0,0.9445009827613831,1.0,0.0,0.744500994682312,0.19358286261558533,0.19358289241790771,0.9445009827613831,1.0,0.0,0.744500994682312,0.19358286261558533,2.0,0.9952805042266846,0.19128578901290894,1.6539866924285889,0.20325925946235657,2026-04-11T15:49:32Z
|
| 33 |
+
0.13171107601374388,0.13171107601374388,0.016304347664117813,0.016304347664117813,0.1480154236778617,0.0,34.0,34.0,26.25,26.25,23.0,23.0,0.9576103650033474,0.001235712079085573,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,32,27.185209274291992,2.1654123663902283,9.69e-06,0.0369,train,60445.0,1.0353732109069824,1.0,0.0,0.8353732228279114,0.3313785493373871,0.3313785493373871,1.0353732109069824,1.0,0.0,0.8353732228279114,0.3313785493373871,2.0,0.9743757247924805,0.26788273453712463,1.3172059059143066,0.1329173594713211,2026-04-11T15:49:36Z
|
| 34 |
+
0.09342801198363304,0.09342801198363304,0.12502007000148296,0.12502007000148296,0.218448081985116,0.0,272.0,272.0,183.75,183.75,112.0,112.0,2.8496975153684616,0.0012743280815569972,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,33,8.614265441894531,1.3449874222278595,9.68e-06,0.0517,train,63579.0,0.5041733980178833,0.75,0.4629100561141968,0.35417336225509644,0.2310473471879959,0.23062638938426971,0.5041733980178833,0.75,0.4629100561141968,0.35417336225509644,0.2310473471879959,2.0,0.9925354719161987,0.12440590560436249,2.0842056274414062,0.197601780295372,2026-04-11T15:49:46Z
|
| 35 |
+
0.0798872783780098,0.0798872783780098,0.08128766575828195,0.08128766575828195,0.16117494413629174,0.0,31.0,31.0,24.875,24.875,13.0,13.0,0.7105261906981468,0.0013129440840284213,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,34,30.032299041748047,2.15890334546566,9.67e-06,0.1286,train,64938.0,0.5193066596984863,1.0,0.0,0.3193066716194153,0.41723787784576416,0.4172378480434418,0.5193066596984863,1.0,0.0,0.3193066716194153,0.41723787784576416,2.0,0.9753797650337219,0.09100068360567093,2.396888256072998,0.1874523013830185,2026-04-11T15:49:51Z
|
| 36 |
+
0.07368538342416286,0.07368538342416286,0.08943439181894064,0.08943439181894064,0.1631197752431035,0.0,65.0,65.0,44.875,44.875,34.0,34.0,2.152104765176773,0.0013515600864998456,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,35,15.259716987609863,1.9157908409833908,9.66e-06,0.1448,train,66585.0,0.7695876359939575,1.0,0.0,0.5695876479148865,0.4217164218425751,0.4217164218425751,0.7695876359939575,1.0,0.0,0.5695876479148865,0.4217164218425751,2.0,1.0017306804656982,0.3279803991317749,1.1148014068603516,0.1511761099100113,2026-04-11T15:49:56Z
|
| 37 |
+
0.11802326329052448,0.11802326329052448,0.059955086559057236,0.059955086559057236,0.17797834984958172,0.0,617.0,617.0,475.125,475.125,352.0,352.0,4.256982505321503,0.0013901760889712697,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,36,4.0301995277404785,0.7125953063368797,9.65e-06,0.1341,train,72074.0,0.7360631227493286,0.25,0.4629100561141968,0.6860631704330444,0.12680380046367645,0.0960330218076706,0.7360631227493286,0.25,0.4629100561141968,0.6860631704330444,0.12680380046367645,2.0,1.0008128881454468,0.2274497151374817,1.4808261394500732,0.13769172132015228,2026-04-11T15:50:14Z
|
| 38 |
+
0.1836030725389719,0.1836030725389719,0.04557228833436966,0.04557228833436966,0.22917536087334156,0.0,83.0,83.0,61.625,61.625,46.0,46.0,3.07864186167717,0.0014287920914426938,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,37,15.153329849243164,1.5459181070327759,9.640000000000001e-06,0.121,train,73927.0,0.8834455609321594,1.0,0.0,0.6834455728530884,0.37012818455696106,0.37012818455696106,0.8834455609321594,1.0,0.0,0.6834455728530884,0.37012818455696106,2.0,0.9819008708000183,0.32161828875541687,1.4087400436401367,0.18970002233982086,2026-04-11T15:50:20Z
|
| 39 |
+
0.10113142617046833,0.10113142617046833,0.042173911817371845,0.042173911817371845,0.14330533798784018,0.0,31.0,31.0,25.75,25.75,23.0,23.0,1.842569574713707,0.001467408093914118,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,38,23.363237380981445,1.7069246247410774,9.630000000000001e-06,0.017,train,75437.0,1.1030879020690918,1.0,0.0,0.903087854385376,0.1558901071548462,0.15589012205600739,1.1030879020690918,1.0,0.0,0.903087854385376,0.1558901071548462,2.0,1.0037627220153809,0.4016261696815491,0.9279258251190186,0.15382516384124756,2026-04-11T15:50:24Z
|
| 40 |
+
0.07904236763715744,0.07904236763715744,0.08287477679550648,0.08287477679550648,0.16191714443266392,0.0,84.0,84.0,74.375,74.375,61.0,61.0,1.1887717097997665,0.0015060240963855422,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,39,18.175195693969727,1.829672947525978,9.620000000000001e-06,0.0859,train,77344.0,0.6274764537811279,1.0,0.0,0.4274764358997345,0.42150962352752686,0.42150962352752686,0.6274764537811279,1.0,0.0,0.4274764358997345,0.42150962352752686,2.0,0.9747925996780396,0.19564573466777802,1.6314496994018555,0.16143742203712463,2026-04-11T15:50:30Z
|
| 41 |
+
0.1951428446918726,0.1951428446918726,0.051804156973958015,0.051804156973958015,0.2469470016658306,0.0,123.0,123.0,89.375,89.375,72.0,72.0,3.59776571393013,0.0015446400988569664,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,40,9.616868019104004,1.3452838510274887,9.610000000000001e-06,0.0531,train,79483.0,1.1042048931121826,1.0,0.0,0.9042049646377563,0.14156009256839752,0.1415601372718811,1.1042048931121826,1.0,0.0,0.9042049646377563,0.14156009256839752,2.0,0.9803248643875122,0.07977253943681717,2.528575897216797,0.1670140027999878,2026-04-11T15:50:36Z
|
| 42 |
+
0.14350228570401669,0.14350228570401669,0.06170275993645191,0.06170275993645191,0.2052050456404686,0.0,365.0,365.0,264.0,264.0,215.0,215.0,3.7767926454544067,0.0015832561013283905,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,41,5.928605556488037,1.0418922081589699,9.600000000000001e-06,0.1829,train,83315.0,0.784883439540863,0.75,0.4629100561141968,0.6348835229873657,0.3371293842792511,0.3861341178417206,0.784883439540863,0.75,0.4629100561141968,0.6348835229873657,0.3371293842792511,2.0,0.9977389574050903,0.3288605809211731,1.3521299362182617,0.15597331523895264,2026-04-11T15:50:48Z
|
| 43 |
+
0.08258083090186119,0.08258083090186119,0.11971818003803492,0.11971818003803492,0.2022990109398961,0.0,88.0,88.0,65.125,65.125,48.0,48.0,3.121804028749466,0.0016218721037998146,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,42,18.00514030456543,2.2616077810525894,9.59e-06,-0.0513,train,85196.0,0.6121174097061157,0.875,0.3535533845424652,0.43711739778518677,0.3802741467952728,0.3986560106277466,0.6121174097061157,0.875,0.3535533845424652,0.43711739778518677,0.3802741467952728,2.0,0.9917340874671936,0.3090456426143646,1.1742663383483887,0.17133519053459167,2026-04-11T15:50:54Z
|
| 44 |
+
0.1285256128758192,0.1285256128758192,0.09967258013784885,0.09967258013784885,0.22819819301366806,0.0,62.0,62.0,47.25,47.25,30.0,30.0,2.3568301051855087,0.0016604881062712389,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,43,16.364044189453125,1.4401512742042542,9.58e-06,-0.0307,train,86814.0,0.6513289213180542,1.0,0.0,0.45132896304130554,0.45054808259010315,0.45054808259010315,0.6513289213180542,1.0,0.0,0.45132896304130554,0.45054808259010315,2.0,0.996967077255249,0.2705899178981781,1.3071508407592773,0.19375590980052948,2026-04-11T15:50:59Z
|
| 45 |
+
0.10046137310564518,0.10046137310564518,0.027404863387346268,0.027404863387346268,0.12786623649299145,0.0,48.0,48.0,43.625,43.625,40.0,40.0,0.9922648519277573,0.001699104108742663,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,44,23.743331909179688,2.507635399699211,9.57e-06,0.0607,train,88539.0,0.8140779733657837,1.0,0.0,0.6140779256820679,0.3976222574710846,0.3976222574710846,0.8140779733657837,1.0,0.0,0.6140779256820679,0.3976222574710846,2.0,0.9870915412902832,0.04806419834494591,3.035217761993408,0.1653001308441162,2026-04-11T15:51:04Z
|
| 46 |
+
0.09030752815306187,0.09030752815306187,0.09415853396058083,0.09415853396058083,0.1844660621136427,0.0,44.0,44.0,38.0,38.0,32.0,32.0,2.8701943159103394,0.001737720111214087,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,45,17.9189510345459,1.6995926201343536,9.56e-06,0.1034,train,90315.0,0.6332358121871948,1.0,0.0,0.433235764503479,0.4090244174003601,0.4090244770050049,0.6332358121871948,1.0,0.0,0.433235764503479,0.4090244174003601,2.0,0.9787829518318176,0.3481438159942627,1.0551395416259766,0.19081313908100128,2026-04-11T15:51:08Z
|
| 47 |
+
0.08075684309005737,0.08075684309005737,0.14126449823379517,0.14126449823379517,0.22202134132385254,0.0,41.0,41.0,29.625,29.625,22.0,22.0,1.8424254208803177,0.0017763361136855112,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,46,35.99632263183594,1.7983057498931885,9.55e-06,0.0292,train,91656.0,0.28713253140449524,1.0,0.0,0.087132528424263,0.08936810493469238,0.08936809748411179,0.28713253140449524,1.0,0.0,0.087132528424263,0.08936810493469238,2.0,0.9919323921203613,0.12090910226106644,2.1127161979675293,0.2213541567325592,2026-04-11T15:51:13Z
|
| 48 |
+
0.0052083334885537624,0.0052083334885537624,0.02500000037252903,0.02500000037252903,0.030208333861082792,0.0,25.0,25.0,24.125,24.125,24.0,24.0,0.34495257679373026,0.0018149521161569355,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,47,36.801151275634766,2.1884380280971527,9.54e-06,0.0968,train,93025.0,1.0150457620620728,1.0,0.0,0.8150457739830017,0.03531856834888458,0.03531854599714279,1.0150457620620728,1.0,0.0,0.8150457739830017,0.03531856834888458,2.0,0.9999908208847046,0.2892707586288452,1.2403922080993652,0.041083987802267075,2026-04-11T15:51:17Z
|
| 49 |
+
0.1315415445715189,0.1315415445715189,0.10152440890669823,0.10152440890669823,0.23306595347821712,0.0,118.0,118.0,87.0,87.0,52.0,52.0,2.6435008347034454,0.0018535681186283596,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,48,17.312498092651367,2.2894017100334167,9.53e-06,0.0324,train,95089.0,0.4331575632095337,1.0,0.0,0.23315757513046265,0.06710986793041229,0.06710987538099289,0.4331575632095337,1.0,0.0,0.23315757513046265,0.06710986793041229,2.0,0.9660635590553284,0.2726844251155853,1.2994401454925537,0.18739104270935059,2026-04-11T15:51:23Z
|
| 50 |
+
0.11550478264689445,0.11550478264689445,0.11420114897191525,0.11420114897191525,0.2297059316188097,0.0,59.0,59.0,43.625,43.625,30.0,30.0,2.8474780917167664,0.0018921841210997837,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,49,15.068387031555176,1.4131104350090027,9.52e-06,0.0945,train,96846.0,0.6715764999389648,1.0,0.0,0.471576452255249,0.4784705936908722,0.4784706234931946,0.6715764999389648,1.0,0.0,0.471576452255249,0.4784705936908722,2.0,1.0054690837860107,0.3757398724555969,1.1681387424468994,0.17449823021888733,2026-04-11T15:51:28Z
|
| 51 |
+
0.14113500528037548,0.14113500528037548,0.07416268065571785,0.07416268065571785,0.21529768593609333,0.0,182.0,182.0,130.5,130.5,88.0,88.0,3.4669112861156464,0.0019308001235712078,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,50,8.942173957824707,1.2031668797135353,9.51e-06,0.0575,train,99306.0,0.6570150256156921,0.875,0.3535533845424652,0.4820150136947632,0.2984078824520111,0.32944732904434204,0.6570150256156921,0.875,0.3535533845424652,0.4820150136947632,0.2984078824520111,2.0,0.9979844093322754,0.2918128967285156,1.5406970977783203,0.167738139629364,2026-04-11T15:51:36Z
|
| 52 |
+
,,,,,,,,,,,,,0.0019308001235712078,0.0,0.0,0.0,0.0,0.0,0.0,320.3076923076923,320.3076923076923,153.05769230769232,153.05769230769232,53.07692307692308,53.07692307692308,3.3071532799647403,0.0,1.2056316297787886,0.018872085958719254,99306.0,0.4456338297862273,0.7307692307692307,0.3936365097761154,0.2994799860394918,0.28621239931537557,0.1990389978656402,0.4456338297862273,0.7307692307692307,0.3936365097761154,0.2994799860394918,0.28621239931537557,217.5742,0.478,1.2880034492566035,0.9996037987562326,0.7560820785852579,0.29305177239271313,0.03544207819952415,0.06,,50,,,,,eval,,,,,,,,,,,,,,,,,,2026-04-11T15:55:14Z
|
| 53 |
+
0.061693549156188965,0.061693549156188965,0.15752424113452435,0.15752424113452435,0.2192177902907133,0.0,43.0,43.0,35.75,35.75,27.0,27.0,2.8815866708755493,0.001969416126042632,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,51,22.3005428314209,1.8204183876514435,9.5e-06,0.1217,train,100856.0,0.46508222818374634,1.0,0.0,0.2650821805000305,0.4003763198852539,0.4003763198852539,0.46508222818374634,1.0,0.0,0.2650821805000305,0.4003763198852539,2.0,0.9820157289505005,0.2877559959888458,1.2456424236297607,0.2505285441875458,2026-04-11T15:55:20Z
|
| 54 |
+
0.16483690962195396,0.16483690962195396,0.013059701770544052,0.013059701770544052,0.17789661139249802,0.0,78.0,78.0,64.875,64.875,40.0,40.0,2.761542499065399,0.002008032128514056,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,,0.0,52,12.406262397766113,1.2337721958756447,9.49e-06,0.0338,train,102767.0,1.1629588603973389,1.0,0.0,0.9629589319229126,0.05814887955784798,0.058148909360170364,1.1629588603973389,1.0,0.0,0.9629589319229126,0.05814887955784798,2.0,1.004599928855896,0.39751237630844116,0.9225292205810547,0.13121938705444336,2026-04-11T15:55:26Z
|
metrics.jsonl
CHANGED
|
@@ -1,5 +1,53 @@
|
|
| 1 |
-
{"timestamp_utc": "2026-04-
|
| 2 |
-
{"timestamp_utc": "2026-04-
|
| 3 |
-
{"timestamp_utc": "2026-04-
|
| 4 |
-
{"timestamp_utc": "2026-04-
|
| 5 |
-
{"timestamp_utc": "2026-04-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"timestamp_utc": "2026-04-11T15:46:43Z", "mode": "train", "global_step": 1, "epoch": 3.861600247142416e-05, "loss": 0.0875, "grad_norm": 14.433236122131348, "learning_rate": 1e-05, "num_tokens": 2073.0, "completions/mean_length": 72.125, "completions/min_length": 68.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.7467405796051025, "rewards/meter/std": 0.2669501304626465, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9467406272888184, "reward_std": 0.26695016026496887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810092180967331, "sampling/sampling_logp_difference/max": 2.3954362869262695, "sampling/importance_sampling_ratio/min": 0.09113290905952454, "sampling/importance_sampling_ratio/mean": 0.9819375872612, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.716715857386589, "entropy": 2.8896071165800095, "clip_ratio/low_mean": 0.09221794456243515, "clip_ratio/low_min": 0.09221794456243515, "clip_ratio/high_mean": 0.16360880248248577, "clip_ratio/high_max": 0.16360880248248577, "clip_ratio/region_mean": 0.2558267470449209, "reward_total_mean": 0.9467406272888184, "reward_meter_mean": 0.7467405796051025, "reward_meter_std": 0.2669501304626465, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 2 |
+
{"timestamp_utc": "2026-04-11T15:46:49Z", "mode": "train", "global_step": 2, "epoch": 7.723200494284832e-05, "loss": 0.1013, "grad_norm": 14.803139686584473, "learning_rate": 9.990000000000001e-06, "num_tokens": 3899.0, "completions/mean_length": 59.25, "completions/min_length": 44.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.692160427570343, "rewards/meter/std": 0.3035436272621155, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8921604156494141, "reward_std": 0.3035435974597931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17282956838607788, "sampling/sampling_logp_difference/max": 1.6599078178405762, "sampling/importance_sampling_ratio/min": 0.27146539092063904, "sampling/importance_sampling_ratio/mean": 0.9748801589012146, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5766338407993317, "entropy": 1.2240338400006294, "clip_ratio/low_mean": 0.037109375, "clip_ratio/low_min": 0.037109375, "clip_ratio/high_mean": 0.15457439702004194, "clip_ratio/high_max": 0.15457439702004194, "clip_ratio/region_mean": 0.19168377202004194, "reward_total_mean": 0.8921604156494141, "reward_meter_mean": 0.692160427570343, "reward_meter_std": 0.3035436272621155, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 3 |
+
{"timestamp_utc": "2026-04-11T15:46:53Z", "mode": "train", "global_step": 3, "epoch": 0.00011584800741427248, "loss": 0.1019, "grad_norm": 22.877466201782227, "learning_rate": 9.980000000000001e-06, "num_tokens": 5552.0, "completions/mean_length": 40.625, "completions/min_length": 32.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5208516120910645, "rewards/meter/std": 0.4123581349849701, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7208516597747803, "reward_std": 0.4123581051826477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18985480070114136, "sampling/sampling_logp_difference/max": 1.3331613540649414, "sampling/importance_sampling_ratio/min": 0.26364248991012573, "sampling/importance_sampling_ratio/mean": 0.9614238142967224, "sampling/importance_sampling_ratio/max": 1.952972412109375, "kl": 2.4038615822792053, "entropy": 2.205263152718544, "clip_ratio/low_mean": 0.08594596944749355, "clip_ratio/low_min": 0.08594596944749355, "clip_ratio/high_mean": 0.11026187054812908, "clip_ratio/high_max": 0.11026187054812908, "clip_ratio/region_mean": 0.19620783999562263, "reward_total_mean": 0.7208516597747803, "reward_meter_mean": 0.5208516120910645, "reward_meter_std": 0.4123581349849701, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 4 |
+
{"timestamp_utc": "2026-04-11T15:46:58Z", "mode": "train", "global_step": 4, "epoch": 0.00015446400988569664, "loss": 0.0914, "grad_norm": 23.66639518737793, "learning_rate": 9.970000000000001e-06, "num_tokens": 6962.0, "completions/mean_length": 22.25, "completions/min_length": 19.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7492154836654663, "rewards/meter/std": 0.4238397777080536, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9492155313491821, "reward_std": 0.4238397479057312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11156373471021652, "sampling/sampling_logp_difference/max": 1.2559912204742432, "sampling/importance_sampling_ratio/min": 0.507404625415802, "sampling/importance_sampling_ratio/mean": 1.0157339572906494, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5658123940229416, "entropy": 1.1889886930584908, "clip_ratio/low_mean": 0.04130434803664684, "clip_ratio/low_min": 0.04130434803664684, "clip_ratio/high_mean": 0.09267808869481087, "clip_ratio/high_max": 0.09267808869481087, "clip_ratio/region_mean": 0.1339824367314577, "reward_total_mean": 0.9492155313491821, "reward_meter_mean": 0.7492154836654663, "reward_meter_std": 0.4238397777080536, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 5 |
+
{"timestamp_utc": "2026-04-11T15:47:07Z", "mode": "train", "global_step": 5, "epoch": 0.0001930800123571208, "loss": -0.0187, "grad_norm": 9.979012489318848, "learning_rate": 9.960000000000001e-06, "num_tokens": 9821.0, "completions/mean_length": 152.375, "completions/min_length": 99.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.375, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.31202924251556396, "rewards/meter/std": 0.2894730269908905, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5120292901992798, "reward_std": 0.2894730269908905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19261623919010162, "sampling/sampling_logp_difference/max": 1.489715576171875, "sampling/importance_sampling_ratio/min": 0.340789794921875, "sampling/importance_sampling_ratio/mean": 0.987079381942749, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.233508050441742, "entropy": 3.3276279270648956, "clip_ratio/low_mean": 0.14424454979598522, "clip_ratio/low_min": 0.14424454979598522, "clip_ratio/high_mean": 0.07602224312722683, "clip_ratio/high_max": 0.07602224312722683, "clip_ratio/region_mean": 0.22026679292321205, "reward_total_mean": 0.5120292901992798, "reward_meter_mean": 0.31202924251556396, "reward_meter_std": 0.2894730269908905, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 6 |
+
{"timestamp_utc": "2026-04-11T15:47:12Z", "mode": "train", "global_step": 6, "epoch": 0.00023169601482854495, "loss": 0.1184, "grad_norm": 15.483458518981934, "learning_rate": 9.950000000000001e-06, "num_tokens": 11697.0, "completions/mean_length": 76.5, "completions/min_length": 61.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.7063730955123901, "rewards/meter/std": 0.21755146980285645, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9063730835914612, "reward_std": 0.21755146980285645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21761518716812134, "sampling/sampling_logp_difference/max": 1.3374531269073486, "sampling/importance_sampling_ratio/min": 0.2625133991241455, "sampling/importance_sampling_ratio/mean": 0.9750673770904541, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.003227472305298, "entropy": 3.393460273742676, "clip_ratio/low_mean": 0.09787485934793949, "clip_ratio/low_min": 0.09787485934793949, "clip_ratio/high_mean": 0.1323312446475029, "clip_ratio/high_max": 0.1323312446475029, "clip_ratio/region_mean": 0.2302061039954424, "reward_total_mean": 0.9063730835914612, "reward_meter_mean": 0.7063730955123901, "reward_meter_std": 0.21755146980285645, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 7 |
+
{"timestamp_utc": "2026-04-11T15:47:19Z", "mode": "train", "global_step": 7, "epoch": 0.0002703120172999691, "loss": 0.2978, "grad_norm": 16.140308380126953, "learning_rate": 9.940000000000001e-06, "num_tokens": 13816.0, "completions/mean_length": 93.875, "completions/min_length": 62.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.4120001196861267, "rewards/meter/std": 0.33185702562332153, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.5870001316070557, "reward_std": 0.36883509159088135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21371272206306458, "sampling/sampling_logp_difference/max": 1.7315394878387451, "sampling/importance_sampling_ratio/min": 0.18673592805862427, "sampling/importance_sampling_ratio/mean": 0.9866839051246643, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.0300692468881607, "entropy": 2.9779615700244904, "clip_ratio/low_mean": 0.12269405275583267, "clip_ratio/low_min": 0.12269405275583267, "clip_ratio/high_mean": 0.07538295164704323, "clip_ratio/high_max": 0.07538295164704323, "clip_ratio/region_mean": 0.1980770044028759, "reward_total_mean": 0.5870001316070557, "reward_meter_mean": 0.4120001196861267, "reward_meter_std": 0.33185702562332153, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 8 |
+
{"timestamp_utc": "2026-04-11T15:47:24Z", "mode": "train", "global_step": 8, "epoch": 0.00030892801977139327, "loss": 0.0345, "grad_norm": 20.1074161529541, "learning_rate": 9.930000000000001e-06, "num_tokens": 15480.0, "completions/mean_length": 42.0, "completions/min_length": 28.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5687531232833862, "rewards/meter/std": 0.3685603737831116, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.7437530755996704, "reward_std": 0.39819100499153137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19006481766700745, "sampling/sampling_logp_difference/max": 1.9432754516601562, "sampling/importance_sampling_ratio/min": 0.1432340294122696, "sampling/importance_sampling_ratio/mean": 0.9912540316581726, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.0594891011714935, "entropy": 2.9420621395111084, "clip_ratio/low_mean": 0.09676836617290974, "clip_ratio/low_min": 0.09676836617290974, "clip_ratio/high_mean": 0.10601851902902126, "clip_ratio/high_max": 0.10601851902902126, "clip_ratio/region_mean": 0.202786885201931, "reward_total_mean": 0.7437530755996704, "reward_meter_mean": 0.5687531232833862, "reward_meter_std": 0.3685603737831116, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 9 |
+
{"timestamp_utc": "2026-04-11T15:47:29Z", "mode": "train", "global_step": 9, "epoch": 0.00034754402224281743, "loss": 0.1313, "grad_norm": 21.095970153808594, "learning_rate": 9.920000000000002e-06, "num_tokens": 17114.0, "completions/mean_length": 47.25, "completions/min_length": 37.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5934403538703918, "rewards/meter/std": 0.4387426972389221, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7934404015541077, "reward_std": 0.4387427270412445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18523243069648743, "sampling/sampling_logp_difference/max": 2.2344932556152344, "sampling/importance_sampling_ratio/min": 0.10704635828733444, "sampling/importance_sampling_ratio/mean": 0.9886912107467651, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5875789374113083, "entropy": 2.3184936344623566, "clip_ratio/low_mean": 0.0800175815820694, "clip_ratio/low_min": 0.0800175815820694, "clip_ratio/high_mean": 0.14036413840949535, "clip_ratio/high_max": 0.14036413840949535, "clip_ratio/region_mean": 0.22038171999156475, "reward_total_mean": 0.7934404015541077, "reward_meter_mean": 0.5934403538703918, "reward_meter_std": 0.4387426972389221, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 10 |
+
{"timestamp_utc": "2026-04-11T15:47:34Z", "mode": "train", "global_step": 10, "epoch": 0.0003861600247142416, "loss": -0.0138, "grad_norm": 17.000186920166016, "learning_rate": 9.91e-06, "num_tokens": 18851.0, "completions/mean_length": 48.125, "completions/min_length": 33.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.4887843132019043, "rewards/meter/std": 0.42971470952033997, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6637843251228333, "reward_std": 0.45756226778030396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2051706463098526, "sampling/sampling_logp_difference/max": 1.3021516799926758, "sampling/importance_sampling_ratio/min": 0.3167512118816376, "sampling/importance_sampling_ratio/mean": 0.9984660148620605, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8380893766880035, "entropy": 3.0384787023067474, "clip_ratio/low_mean": 0.16380306333303452, "clip_ratio/low_min": 0.16380306333303452, "clip_ratio/high_mean": 0.09042712114751339, "clip_ratio/high_max": 0.09042712114751339, "clip_ratio/region_mean": 0.2542301844805479, "reward_total_mean": 0.6637843251228333, "reward_meter_mean": 0.4887843132019043, "reward_meter_std": 0.42971470952033997, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 11 |
+
{"timestamp_utc": "2026-04-11T15:47:39Z", "mode": "train", "global_step": 11, "epoch": 0.00042477602718566575, "loss": 0.1831, "grad_norm": 16.804243087768555, "learning_rate": 9.9e-06, "num_tokens": 20578.0, "completions/mean_length": 44.875, "completions/min_length": 29.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.30691099166870117, "rewards/meter/std": 0.4142797291278839, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.506911039352417, "reward_std": 0.4142796993255615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18619823455810547, "sampling/sampling_logp_difference/max": 1.9143197536468506, "sampling/importance_sampling_ratio/min": 0.14744210243225098, "sampling/importance_sampling_ratio/mean": 0.9988991618156433, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.717022493481636, "entropy": 2.5153299272060394, "clip_ratio/low_mean": 0.11223171092569828, "clip_ratio/low_min": 0.11223171092569828, "clip_ratio/high_mean": 0.08737440872937441, "clip_ratio/high_max": 0.08737440872937441, "clip_ratio/region_mean": 0.1996061196550727, "reward_total_mean": 0.506911039352417, "reward_meter_mean": 0.30691099166870117, "reward_meter_std": 0.4142797291278839, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 12 |
+
{"timestamp_utc": "2026-04-11T15:47:44Z", "mode": "train", "global_step": 12, "epoch": 0.0004633920296570899, "loss": -0.02, "grad_norm": 22.950233459472656, "learning_rate": 9.89e-06, "num_tokens": 22530.0, "completions/mean_length": 58.0, "completions/min_length": 38.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.2489539384841919, "rewards/meter/std": 0.23509462177753448, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4489539563655853, "reward_std": 0.23509462177753448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2930803596973419, "sampling/sampling_logp_difference/max": 3.918379306793213, "sampling/importance_sampling_ratio/min": 0.07527641206979752, "sampling/importance_sampling_ratio/mean": 0.9429361820220947, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.302564397454262, "entropy": 1.546494573354721, "clip_ratio/low_mean": 0.13612079434096813, "clip_ratio/low_min": 0.13612079434096813, "clip_ratio/high_mean": 0.048309178091585636, "clip_ratio/high_max": 0.048309178091585636, "clip_ratio/region_mean": 0.18442997243255377, "reward_total_mean": 0.4489539563655853, "reward_meter_mean": 0.2489539384841919, "reward_meter_std": 0.23509462177753448, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 13 |
+
{"timestamp_utc": "2026-04-11T15:47:49Z", "mode": "train", "global_step": 13, "epoch": 0.000502008032128514, "loss": 0.0551, "grad_norm": 20.975563049316406, "learning_rate": 9.88e-06, "num_tokens": 24249.0, "completions/mean_length": 46.875, "completions/min_length": 33.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5702009201049805, "rewards/meter/std": 0.416922926902771, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7702009677886963, "reward_std": 0.41692298650741577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19165004789829254, "sampling/sampling_logp_difference/max": 1.6037216186523438, "sampling/importance_sampling_ratio/min": 0.20114654302597046, "sampling/importance_sampling_ratio/mean": 0.9821399450302124, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.4146449714899063, "entropy": 2.7278781831264496, "clip_ratio/low_mean": 0.0789086427539587, "clip_ratio/low_min": 0.0789086427539587, "clip_ratio/high_mean": 0.11046412773430347, "clip_ratio/high_max": 0.11046412773430347, "clip_ratio/region_mean": 0.18937277048826218, "reward_total_mean": 0.7702009677886963, "reward_meter_mean": 0.5702009201049805, "reward_meter_std": 0.416922926902771, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 14 |
+
{"timestamp_utc": "2026-04-11T15:47:56Z", "mode": "train", "global_step": 14, "epoch": 0.0005406240345999382, "loss": 0.0636, "grad_norm": 13.641975402832031, "learning_rate": 9.87e-06, "num_tokens": 26453.0, "completions/mean_length": 82.5, "completions/min_length": 63.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.3489101529121399, "rewards/meter/std": 0.34769579768180847, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5489101409912109, "reward_std": 0.34769579768180847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2145778238773346, "sampling/sampling_logp_difference/max": 2.2805795669555664, "sampling/importance_sampling_ratio/min": 0.1022249385714531, "sampling/importance_sampling_ratio/mean": 0.9998252987861633, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3619689494371414, "entropy": 2.838075205683708, "clip_ratio/low_mean": 0.13307647220790386, "clip_ratio/low_min": 0.13307647220790386, "clip_ratio/high_mean": 0.09134172648191452, "clip_ratio/high_max": 0.09134172648191452, "clip_ratio/region_mean": 0.22441819868981838, "reward_total_mean": 0.5489101409912109, "reward_meter_mean": 0.3489101529121399, "reward_meter_std": 0.34769579768180847, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 15 |
+
{"timestamp_utc": "2026-04-11T15:48:00Z", "mode": "train", "global_step": 15, "epoch": 0.0005792400370713623, "loss": 0.1104, "grad_norm": 26.442886352539062, "learning_rate": 9.86e-06, "num_tokens": 27811.0, "completions/mean_length": 22.75, "completions/min_length": 15.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6123703718185425, "rewards/meter/std": 0.42484015226364136, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8123704195022583, "reward_std": 0.42484015226364136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21215009689331055, "sampling/sampling_logp_difference/max": 2.0358657836914062, "sampling/importance_sampling_ratio/min": 0.3356194794178009, "sampling/importance_sampling_ratio/mean": 0.9939978122711182, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7645654305815697, "entropy": 2.376413881778717, "clip_ratio/low_mean": 0.05330882500857115, "clip_ratio/low_min": 0.05330882500857115, "clip_ratio/high_mean": 0.11765297874808311, "clip_ratio/high_max": 0.11765297874808311, "clip_ratio/region_mean": 0.17096180375665426, "reward_total_mean": 0.8123704195022583, "reward_meter_mean": 0.6123703718185425, "reward_meter_std": 0.42484015226364136, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 16 |
+
{"timestamp_utc": "2026-04-11T15:48:05Z", "mode": "train", "global_step": 16, "epoch": 0.0006178560395427865, "loss": 0.105, "grad_norm": 20.25496482849121, "learning_rate": 9.85e-06, "num_tokens": 29402.0, "completions/mean_length": 34.875, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.59489905834198, "rewards/meter/std": 0.4257949888706207, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7948991060256958, "reward_std": 0.42579495906829834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2184409201145172, "sampling/sampling_logp_difference/max": 1.4742670059204102, "sampling/importance_sampling_ratio/min": 0.24731610715389252, "sampling/importance_sampling_ratio/mean": 0.967242956161499, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.90836800634861, "entropy": 2.617787539958954, "clip_ratio/low_mean": 0.10445018857717514, "clip_ratio/low_min": 0.10445018857717514, "clip_ratio/high_mean": 0.10188646428287029, "clip_ratio/high_max": 0.10188646428287029, "clip_ratio/region_mean": 0.20633665286004543, "reward_total_mean": 0.7948991060256958, "reward_meter_mean": 0.59489905834198, "reward_meter_std": 0.4257949888706207, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 17 |
+
{"timestamp_utc": "2026-04-11T15:48:09Z", "mode": "train", "global_step": 17, "epoch": 0.0006564720420142106, "loss": 0.1285, "grad_norm": 25.74098014831543, "learning_rate": 9.84e-06, "num_tokens": 30963.0, "completions/mean_length": 41.125, "completions/min_length": 26.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8503542542457581, "rewards/meter/std": 0.3227464556694031, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.050354242324829, "reward_std": 0.32274648547172546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10933797806501389, "sampling/sampling_logp_difference/max": 1.4986724853515625, "sampling/importance_sampling_ratio/min": 0.22342658042907715, "sampling/importance_sampling_ratio/mean": 0.9962012767791748, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7420838475227356, "entropy": 0.8714146893471479, "clip_ratio/low_mean": 0.059752749279141426, "clip_ratio/low_min": 0.059752749279141426, "clip_ratio/high_mean": 0.056653511710464954, "clip_ratio/high_max": 0.056653511710464954, "clip_ratio/region_mean": 0.11640626098960638, "reward_total_mean": 1.050354242324829, "reward_meter_mean": 0.8503542542457581, "reward_meter_std": 0.3227464556694031, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 18 |
+
{"timestamp_utc": "2026-04-11T15:48:14Z", "mode": "train", "global_step": 18, "epoch": 0.0006950880444856349, "loss": 0.1207, "grad_norm": 17.443212509155273, "learning_rate": 9.83e-06, "num_tokens": 32520.0, "completions/mean_length": 42.625, "completions/min_length": 31.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.31625181436538696, "rewards/meter/std": 0.2878619432449341, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5162518620491028, "reward_std": 0.2878619432449341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1679675132036209, "sampling/sampling_logp_difference/max": 0.9356918334960938, "sampling/importance_sampling_ratio/min": 0.3923143446445465, "sampling/importance_sampling_ratio/mean": 0.9975839257240295, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4733051806688309, "entropy": 2.5996862947940826, "clip_ratio/low_mean": 0.11914125084877014, "clip_ratio/low_min": 0.11914125084877014, "clip_ratio/high_mean": 0.07529962994158268, "clip_ratio/high_max": 0.07529962994158268, "clip_ratio/region_mean": 0.19444088079035282, "reward_total_mean": 0.5162518620491028, "reward_meter_mean": 0.31625181436538696, "reward_meter_std": 0.2878619432449341, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 19 |
+
{"timestamp_utc": "2026-04-11T15:48:22Z", "mode": "train", "global_step": 19, "epoch": 0.000733704046957059, "loss": 0.1359, "grad_norm": 8.90630054473877, "learning_rate": 9.820000000000001e-06, "num_tokens": 34880.0, "completions/mean_length": 119.0, "completions/min_length": 67.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.4886695444583893, "rewards/meter/std": 0.21834495663642883, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.6386695504188538, "reward_std": 0.25513508915901184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16941869258880615, "sampling/sampling_logp_difference/max": 1.298689365386963, "sampling/importance_sampling_ratio/min": 0.27288922667503357, "sampling/importance_sampling_ratio/mean": 1.001911997795105, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0056203603744507, "entropy": 2.8221229016780853, "clip_ratio/low_mean": 0.11502809636294842, "clip_ratio/low_min": 0.11502809636294842, "clip_ratio/high_mean": 0.08858394995331764, "clip_ratio/high_max": 0.08858394995331764, "clip_ratio/region_mean": 0.20361204631626606, "reward_total_mean": 0.6386695504188538, "reward_meter_mean": 0.4886695444583893, "reward_meter_std": 0.21834495663642883, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968}
|
| 20 |
+
{"timestamp_utc": "2026-04-11T15:48:27Z", "mode": "train", "global_step": 20, "epoch": 0.0007723200494284832, "loss": 0.1452, "grad_norm": 14.272367477416992, "learning_rate": 9.810000000000001e-06, "num_tokens": 36697.0, "completions/mean_length": 53.125, "completions/min_length": 35.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.584618330001831, "rewards/meter/std": 0.37843555212020874, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7846182584762573, "reward_std": 0.37843549251556396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18555279076099396, "sampling/sampling_logp_difference/max": 1.712082862854004, "sampling/importance_sampling_ratio/min": 0.1804894655942917, "sampling/importance_sampling_ratio/mean": 0.9827072620391846, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.500285528600216, "entropy": 2.9686510264873505, "clip_ratio/low_mean": 0.1128854975104332, "clip_ratio/low_min": 0.1128854975104332, "clip_ratio/high_mean": 0.12786824442446232, "clip_ratio/high_max": 0.12786824442446232, "clip_ratio/region_mean": 0.24075374193489552, "reward_total_mean": 0.7846182584762573, "reward_meter_mean": 0.584618330001831, "reward_meter_std": 0.37843555212020874, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 21 |
+
{"timestamp_utc": "2026-04-11T15:48:34Z", "mode": "train", "global_step": 21, "epoch": 0.0008109360518999073, "loss": 0.0602, "grad_norm": 8.9111328125, "learning_rate": 9.800000000000001e-06, "num_tokens": 39216.0, "completions/mean_length": 127.875, "completions/min_length": 88.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.5299741625785828, "rewards/meter/std": 0.38676267862319946, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7299741506576538, "reward_std": 0.38676267862319946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1731364130973816, "sampling/sampling_logp_difference/max": 1.4290367364883423, "sampling/importance_sampling_ratio/min": 0.23953956365585327, "sampling/importance_sampling_ratio/mean": 0.9976959228515625, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3634046390652657, "entropy": 3.330483376979828, "clip_ratio/low_mean": 0.1339168418198824, "clip_ratio/low_min": 0.1339168418198824, "clip_ratio/high_mean": 0.08470636792480946, "clip_ratio/high_max": 0.08470636792480946, "clip_ratio/region_mean": 0.21862320974469185, "reward_total_mean": 0.7299741506576538, "reward_meter_mean": 0.5299741625785828, "reward_meter_std": 0.38676267862319946, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 22 |
+
{"timestamp_utc": "2026-04-11T15:48:39Z", "mode": "train", "global_step": 22, "epoch": 0.0008495520543713315, "loss": 0.0956, "grad_norm": 19.450031280517578, "learning_rate": 9.790000000000001e-06, "num_tokens": 40839.0, "completions/mean_length": 41.875, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7068014740943909, "rewards/meter/std": 0.39332520961761475, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9068014621734619, "reward_std": 0.39332520961761475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19435128569602966, "sampling/sampling_logp_difference/max": 1.2685363292694092, "sampling/importance_sampling_ratio/min": 0.2812429666519165, "sampling/importance_sampling_ratio/mean": 0.9838151931762695, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.744483582675457, "entropy": 1.974789410829544, "clip_ratio/low_mean": 0.09037221781909466, "clip_ratio/low_min": 0.09037221781909466, "clip_ratio/high_mean": 0.11499525140970945, "clip_ratio/high_max": 0.11499525140970945, "clip_ratio/region_mean": 0.2053674692288041, "reward_total_mean": 0.9068014621734619, "reward_meter_mean": 0.7068014740943909, "reward_meter_std": 0.39332520961761475, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 23 |
+
{"timestamp_utc": "2026-04-11T15:48:44Z", "mode": "train", "global_step": 23, "epoch": 0.0008881680568427556, "loss": -0.0134, "grad_norm": 11.986440658569336, "learning_rate": 9.780000000000001e-06, "num_tokens": 42619.0, "completions/mean_length": 59.5, "completions/min_length": 51.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.3556836247444153, "rewards/meter/std": 0.41377386450767517, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5556836128234863, "reward_std": 0.4137738347053528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15584221482276917, "sampling/sampling_logp_difference/max": 0.9905836582183838, "sampling/importance_sampling_ratio/min": 0.37629470229148865, "sampling/importance_sampling_ratio/mean": 0.9997124671936035, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2204247117042542, "entropy": 2.614344671368599, "clip_ratio/low_mean": 0.12430914491415024, "clip_ratio/low_min": 0.12430914491415024, "clip_ratio/high_mean": 0.0615717563778162, "clip_ratio/high_max": 0.0615717563778162, "clip_ratio/region_mean": 0.18588090129196644, "reward_total_mean": 0.5556836128234863, "reward_meter_mean": 0.3556836247444153, "reward_meter_std": 0.41377386450767517, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 24 |
+
{"timestamp_utc": "2026-04-11T15:48:49Z", "mode": "train", "global_step": 24, "epoch": 0.0009267840593141798, "loss": 0.0328, "grad_norm": 15.669774055480957, "learning_rate": 9.770000000000001e-06, "num_tokens": 44409.0, "completions/mean_length": 54.75, "completions/min_length": 29.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5161629915237427, "rewards/meter/std": 0.41966450214385986, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6911630034446716, "reward_std": 0.40840157866477966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1744653582572937, "sampling/sampling_logp_difference/max": 1.210925817489624, "sampling/importance_sampling_ratio/min": 0.2979213297367096, "sampling/importance_sampling_ratio/mean": 1.0036118030548096, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.9536623805761337, "entropy": 2.8385555744171143, "clip_ratio/low_mean": 0.10080079734325409, "clip_ratio/low_min": 0.10080079734325409, "clip_ratio/high_mean": 0.08839947171509266, "clip_ratio/high_max": 0.08839947171509266, "clip_ratio/region_mean": 0.18920026905834675, "reward_total_mean": 0.6911630034446716, "reward_meter_mean": 0.5161629915237427, "reward_meter_std": 0.41966450214385986, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 25 |
+
{"timestamp_utc": "2026-04-11T15:48:54Z", "mode": "train", "global_step": 25, "epoch": 0.0009654000617856039, "loss": 0.056, "grad_norm": 13.490758895874023, "learning_rate": 9.760000000000001e-06, "num_tokens": 46042.0, "completions/mean_length": 36.125, "completions/min_length": 25.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.2736630439758301, "rewards/meter/std": 0.30189424753189087, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4736630320549011, "reward_std": 0.3018941879272461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13724787533283234, "sampling/sampling_logp_difference/max": 1.1764202117919922, "sampling/importance_sampling_ratio/min": 0.3343762159347534, "sampling/importance_sampling_ratio/mean": 0.9891919493675232, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5680749863386154, "entropy": 1.6488263756036758, "clip_ratio/low_mean": 0.12277928367257118, "clip_ratio/low_min": 0.12277928367257118, "clip_ratio/high_mean": 0.035714286379516125, "clip_ratio/high_max": 0.035714286379516125, "clip_ratio/region_mean": 0.1584935700520873, "reward_total_mean": 0.4736630320549011, "reward_meter_mean": 0.2736630439758301, "reward_meter_std": 0.30189424753189087, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 26 |
+
{"timestamp_utc": "2026-04-11T15:49:00Z", "mode": "train", "global_step": 26, "epoch": 0.001004016064257028, "loss": 0.202, "grad_norm": 16.0242919921875, "learning_rate": 9.75e-06, "num_tokens": 47681.0, "completions/mean_length": 44.875, "completions/min_length": 28.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.42073720693588257, "rewards/meter/std": 0.3384658396244049, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6207371950149536, "reward_std": 0.3384658098220825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17049100995063782, "sampling/sampling_logp_difference/max": 1.2931162118911743, "sampling/importance_sampling_ratio/min": 0.2744143009185791, "sampling/importance_sampling_ratio/mean": 0.9966971278190613, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.6573083698749542, "entropy": 2.3349276185035706, "clip_ratio/low_mean": 0.10312438476830721, "clip_ratio/low_min": 0.10312438476830721, "clip_ratio/high_mean": 0.07951899617910385, "clip_ratio/high_max": 0.07951899617910385, "clip_ratio/region_mean": 0.18264338094741106, "reward_total_mean": 0.6207371950149536, "reward_meter_mean": 0.42073720693588257, "reward_meter_std": 0.3384658396244049, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 27 |
+
{"timestamp_utc": "2026-04-11T15:49:07Z", "mode": "train", "global_step": 27, "epoch": 0.0010426320667284523, "loss": 0.1939, "grad_norm": 21.260799407958984, "learning_rate": 9.74e-06, "num_tokens": 50118.0, "completions/mean_length": 114.625, "completions/min_length": 79.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.42070716619491577, "rewards/meter/std": 0.32767125964164734, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.5957071781158447, "reward_std": 0.3487240672111511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19498828053474426, "sampling/sampling_logp_difference/max": 1.8702855110168457, "sampling/importance_sampling_ratio/min": 0.15407966077327728, "sampling/importance_sampling_ratio/mean": 0.9976061582565308, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.746354952454567, "entropy": 2.8878106623888016, "clip_ratio/low_mean": 0.11959273181855679, "clip_ratio/low_min": 0.11959273181855679, "clip_ratio/high_mean": 0.11580519005656242, "clip_ratio/high_max": 0.11580519005656242, "clip_ratio/region_mean": 0.2353979218751192, "reward_total_mean": 0.5957071781158447, "reward_meter_mean": 0.42070716619491577, "reward_meter_std": 0.32767125964164734, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 28 |
+
{"timestamp_utc": "2026-04-11T15:49:13Z", "mode": "train", "global_step": 28, "epoch": 0.0010812480691998764, "loss": 0.1229, "grad_norm": 25.492298126220703, "learning_rate": 9.73e-06, "num_tokens": 52550.0, "completions/mean_length": 102.0, "completions/min_length": 86.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.4947388172149658, "rewards/meter/std": 0.2714572250843048, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6697388291358948, "reward_std": 0.23900270462036133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2764182388782501, "sampling/sampling_logp_difference/max": 2.778400421142578, "sampling/importance_sampling_ratio/min": 0.062137819826602936, "sampling/importance_sampling_ratio/mean": 0.9748236536979675, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.430370479822159, "entropy": 1.6325490325689316, "clip_ratio/low_mean": 0.12361728027462959, "clip_ratio/low_min": 0.12361728027462959, "clip_ratio/high_mean": 0.0886442456394434, "clip_ratio/high_max": 0.0886442456394434, "clip_ratio/region_mean": 0.212261525914073, "reward_total_mean": 0.6697388291358948, "reward_meter_mean": 0.4947388172149658, "reward_meter_std": 0.2714572250843048, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 29 |
+
{"timestamp_utc": "2026-04-11T15:49:18Z", "mode": "train", "global_step": 29, "epoch": 0.0011198640716713006, "loss": -0.0165, "grad_norm": 23.189964294433594, "learning_rate": 9.72e-06, "num_tokens": 54344.0, "completions/mean_length": 43.25, "completions/min_length": 34.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6811927556991577, "rewards/meter/std": 0.32601746916770935, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8811927437782288, "reward_std": 0.3260175287723541, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18413126468658447, "sampling/sampling_logp_difference/max": 1.2698702812194824, "sampling/importance_sampling_ratio/min": 0.2808680534362793, "sampling/importance_sampling_ratio/mean": 1.0205129384994507, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8857728093862534, "entropy": 2.0946053713560104, "clip_ratio/low_mean": 0.07247674837708473, "clip_ratio/low_min": 0.07247674837708473, "clip_ratio/high_mean": 0.15308464504778385, "clip_ratio/high_max": 0.15308464504778385, "clip_ratio/region_mean": 0.22556139342486858, "reward_total_mean": 0.8811927437782288, "reward_meter_mean": 0.6811927556991577, "reward_meter_std": 0.32601746916770935, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 30 |
+
{"timestamp_utc": "2026-04-11T15:49:27Z", "mode": "train", "global_step": 30, "epoch": 0.0011584800741427247, "loss": -0.1499, "grad_norm": 9.213912963867188, "learning_rate": 9.71e-06, "num_tokens": 57126.0, "completions/mean_length": 146.75, "completions/min_length": 71.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.3599656820297241, "rewards/meter/std": 0.22725822031497955, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5599656701087952, "reward_std": 0.22725820541381836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17979256808757782, "sampling/sampling_logp_difference/max": 1.3972516059875488, "sampling/importance_sampling_ratio/min": 0.24727565050125122, "sampling/importance_sampling_ratio/mean": 1.0062932968139648, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0472772866487503, "entropy": 3.4500816762447357, "clip_ratio/low_mean": 0.093433802947402, "clip_ratio/low_min": 0.093433802947402, "clip_ratio/high_mean": 0.1298079490661621, "clip_ratio/high_max": 0.1298079490661621, "clip_ratio/region_mean": 0.2232417520135641, "reward_total_mean": 0.5599656701087952, "reward_meter_mean": 0.3599656820297241, "reward_meter_std": 0.22725822031497955, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 31 |
+
{"timestamp_utc": "2026-04-11T15:49:32Z", "mode": "train", "global_step": 31, "epoch": 0.001197096076614149, "loss": 0.0747, "grad_norm": 20.018091201782227, "learning_rate": 9.7e-06, "num_tokens": 58875.0, "completions/mean_length": 56.625, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.744500994682312, "rewards/meter/std": 0.19358286261558533, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9445009827613831, "reward_std": 0.19358289241790771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20325925946235657, "sampling/sampling_logp_difference/max": 1.6539866924285889, "sampling/importance_sampling_ratio/min": 0.19128578901290894, "sampling/importance_sampling_ratio/mean": 0.9952805042266846, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5809837579727173, "entropy": 2.848589450120926, "clip_ratio/low_mean": 0.12893840484321117, "clip_ratio/low_min": 0.12893840484321117, "clip_ratio/high_mean": 0.09737317077815533, "clip_ratio/high_max": 0.09737317077815533, "clip_ratio/region_mean": 0.2263115756213665, "reward_total_mean": 0.9445009827613831, "reward_meter_mean": 0.744500994682312, "reward_meter_std": 0.19358286261558533, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 32 |
+
{"timestamp_utc": "2026-04-11T15:49:36Z", "mode": "train", "global_step": 32, "epoch": 0.001235712079085573, "loss": 0.0369, "grad_norm": 27.185209274291992, "learning_rate": 9.69e-06, "num_tokens": 60445.0, "completions/mean_length": 26.25, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8353732228279114, "rewards/meter/std": 0.3313785493373871, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.0353732109069824, "reward_std": 0.3313785493373871, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1329173594713211, "sampling/sampling_logp_difference/max": 1.3172059059143066, "sampling/importance_sampling_ratio/min": 0.26788273453712463, "sampling/importance_sampling_ratio/mean": 0.9743757247924805, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.1654123663902283, "entropy": 0.9576103650033474, "clip_ratio/low_mean": 0.016304347664117813, "clip_ratio/low_min": 0.016304347664117813, "clip_ratio/high_mean": 0.13171107601374388, "clip_ratio/high_max": 0.13171107601374388, "clip_ratio/region_mean": 0.1480154236778617, "reward_total_mean": 1.0353732109069824, "reward_meter_mean": 0.8353732228279114, "reward_meter_std": 0.3313785493373871, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 33 |
+
{"timestamp_utc": "2026-04-11T15:49:46Z", "mode": "train", "global_step": 33, "epoch": 0.0012743280815569972, "loss": 0.0517, "grad_norm": 8.614265441894531, "learning_rate": 9.68e-06, "num_tokens": 63579.0, "completions/mean_length": 183.75, "completions/min_length": 112.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.35417336225509644, "rewards/meter/std": 0.2310473471879959, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.5041733980178833, "reward_std": 0.23062638938426971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.197601780295372, "sampling/sampling_logp_difference/max": 2.0842056274414062, "sampling/importance_sampling_ratio/min": 0.12440590560436249, "sampling/importance_sampling_ratio/mean": 0.9925354719161987, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3449874222278595, "entropy": 2.8496975153684616, "clip_ratio/low_mean": 0.12502007000148296, "clip_ratio/low_min": 0.12502007000148296, "clip_ratio/high_mean": 0.09342801198363304, "clip_ratio/high_max": 0.09342801198363304, "clip_ratio/region_mean": 0.218448081985116, "reward_total_mean": 0.5041733980178833, "reward_meter_mean": 0.35417336225509644, "reward_meter_std": 0.2310473471879959, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968}
|
| 34 |
+
{"timestamp_utc": "2026-04-11T15:49:51Z", "mode": "train", "global_step": 34, "epoch": 0.0013129440840284213, "loss": 0.1286, "grad_norm": 30.032299041748047, "learning_rate": 9.67e-06, "num_tokens": 64938.0, "completions/mean_length": 24.875, "completions/min_length": 13.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.3193066716194153, "rewards/meter/std": 0.41723787784576416, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5193066596984863, "reward_std": 0.4172378480434418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1874523013830185, "sampling/sampling_logp_difference/max": 2.396888256072998, "sampling/importance_sampling_ratio/min": 0.09100068360567093, "sampling/importance_sampling_ratio/mean": 0.9753797650337219, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.15890334546566, "entropy": 0.7105261906981468, "clip_ratio/low_mean": 0.08128766575828195, "clip_ratio/low_min": 0.08128766575828195, "clip_ratio/high_mean": 0.0798872783780098, "clip_ratio/high_max": 0.0798872783780098, "clip_ratio/region_mean": 0.16117494413629174, "reward_total_mean": 0.5193066596984863, "reward_meter_mean": 0.3193066716194153, "reward_meter_std": 0.41723787784576416, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 35 |
+
{"timestamp_utc": "2026-04-11T15:49:56Z", "mode": "train", "global_step": 35, "epoch": 0.0013515600864998456, "loss": 0.1448, "grad_norm": 15.259716987609863, "learning_rate": 9.66e-06, "num_tokens": 66585.0, "completions/mean_length": 44.875, "completions/min_length": 34.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5695876479148865, "rewards/meter/std": 0.4217164218425751, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7695876359939575, "reward_std": 0.4217164218425751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1511761099100113, "sampling/sampling_logp_difference/max": 1.1148014068603516, "sampling/importance_sampling_ratio/min": 0.3279803991317749, "sampling/importance_sampling_ratio/mean": 1.0017306804656982, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.9157908409833908, "entropy": 2.152104765176773, "clip_ratio/low_mean": 0.08943439181894064, "clip_ratio/low_min": 0.08943439181894064, "clip_ratio/high_mean": 0.07368538342416286, "clip_ratio/high_max": 0.07368538342416286, "clip_ratio/region_mean": 0.1631197752431035, "reward_total_mean": 0.7695876359939575, "reward_meter_mean": 0.5695876479148865, "reward_meter_std": 0.4217164218425751, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 36 |
+
{"timestamp_utc": "2026-04-11T15:50:14Z", "mode": "train", "global_step": 36, "epoch": 0.0013901760889712697, "loss": 0.1341, "grad_norm": 4.0301995277404785, "learning_rate": 9.65e-06, "num_tokens": 72074.0, "completions/mean_length": 475.125, "completions/min_length": 352.0, "completions/max_length": 617.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 475.125, "completions/min_terminated_length": 352.0, "completions/max_terminated_length": 617.0, "rewards/meter/mean": 0.6860631704330444, "rewards/meter/std": 0.12680380046367645, "rewards/exact_count_bonus/mean": 0.25, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.7360631227493286, "reward_std": 0.0960330218076706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13769172132015228, "sampling/sampling_logp_difference/max": 1.4808261394500732, "sampling/importance_sampling_ratio/min": 0.2274497151374817, "sampling/importance_sampling_ratio/mean": 1.0008128881454468, "sampling/importance_sampling_ratio/max": 2.0, "kl": 0.7125953063368797, "entropy": 4.256982505321503, "clip_ratio/low_mean": 0.059955086559057236, "clip_ratio/low_min": 0.059955086559057236, "clip_ratio/high_mean": 0.11802326329052448, "clip_ratio/high_max": 0.11802326329052448, "clip_ratio/region_mean": 0.17797834984958172, "reward_total_mean": 0.7360631227493286, "reward_meter_mean": 0.6860631704330444, "reward_meter_std": 0.12680380046367645, "reward_exact_count_bonus_mean": 0.25, "reward_exact_count_bonus_std": 0.4629100561141968}
|
| 37 |
+
{"timestamp_utc": "2026-04-11T15:50:20Z", "mode": "train", "global_step": 37, "epoch": 0.0014287920914426938, "loss": 0.121, "grad_norm": 15.153329849243164, "learning_rate": 9.640000000000001e-06, "num_tokens": 73927.0, "completions/mean_length": 61.625, "completions/min_length": 46.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6834455728530884, "rewards/meter/std": 0.37012818455696106, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8834455609321594, "reward_std": 0.37012818455696106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18970002233982086, "sampling/sampling_logp_difference/max": 1.4087400436401367, "sampling/importance_sampling_ratio/min": 0.32161828875541687, "sampling/importance_sampling_ratio/mean": 0.9819008708000183, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5459181070327759, "entropy": 3.07864186167717, "clip_ratio/low_mean": 0.04557228833436966, "clip_ratio/low_min": 0.04557228833436966, "clip_ratio/high_mean": 0.1836030725389719, "clip_ratio/high_max": 0.1836030725389719, "clip_ratio/region_mean": 0.22917536087334156, "reward_total_mean": 0.8834455609321594, "reward_meter_mean": 0.6834455728530884, "reward_meter_std": 0.37012818455696106, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 38 |
+
{"timestamp_utc": "2026-04-11T15:50:24Z", "mode": "train", "global_step": 38, "epoch": 0.001467408093914118, "loss": 0.017, "grad_norm": 23.363237380981445, "learning_rate": 9.630000000000001e-06, "num_tokens": 75437.0, "completions/mean_length": 25.75, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.903087854385376, "rewards/meter/std": 0.1558901071548462, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1030879020690918, "reward_std": 0.15589012205600739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15382516384124756, "sampling/sampling_logp_difference/max": 0.9279258251190186, "sampling/importance_sampling_ratio/min": 0.4016261696815491, "sampling/importance_sampling_ratio/mean": 1.0037627220153809, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7069246247410774, "entropy": 1.842569574713707, "clip_ratio/low_mean": 0.042173911817371845, "clip_ratio/low_min": 0.042173911817371845, "clip_ratio/high_mean": 0.10113142617046833, "clip_ratio/high_max": 0.10113142617046833, "clip_ratio/region_mean": 0.14330533798784018, "reward_total_mean": 1.1030879020690918, "reward_meter_mean": 0.903087854385376, "reward_meter_std": 0.1558901071548462, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 39 |
+
{"timestamp_utc": "2026-04-11T15:50:30Z", "mode": "train", "global_step": 39, "epoch": 0.0015060240963855422, "loss": 0.0859, "grad_norm": 18.175195693969727, "learning_rate": 9.620000000000001e-06, "num_tokens": 77344.0, "completions/mean_length": 74.375, "completions/min_length": 61.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.4274764358997345, "rewards/meter/std": 0.42150962352752686, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6274764537811279, "reward_std": 0.42150962352752686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16143742203712463, "sampling/sampling_logp_difference/max": 1.6314496994018555, "sampling/importance_sampling_ratio/min": 0.19564573466777802, "sampling/importance_sampling_ratio/mean": 0.9747925996780396, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.829672947525978, "entropy": 1.1887717097997665, "clip_ratio/low_mean": 0.08287477679550648, "clip_ratio/low_min": 0.08287477679550648, "clip_ratio/high_mean": 0.07904236763715744, "clip_ratio/high_max": 0.07904236763715744, "clip_ratio/region_mean": 0.16191714443266392, "reward_total_mean": 0.6274764537811279, "reward_meter_mean": 0.4274764358997345, "reward_meter_std": 0.42150962352752686, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 40 |
+
{"timestamp_utc": "2026-04-11T15:50:36Z", "mode": "train", "global_step": 40, "epoch": 0.0015446400988569664, "loss": 0.0531, "grad_norm": 9.616868019104004, "learning_rate": 9.610000000000001e-06, "num_tokens": 79483.0, "completions/mean_length": 89.375, "completions/min_length": 72.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9042049646377563, "rewards/meter/std": 0.14156009256839752, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1042048931121826, "reward_std": 0.1415601372718811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1670140027999878, "sampling/sampling_logp_difference/max": 2.528575897216797, "sampling/importance_sampling_ratio/min": 0.07977253943681717, "sampling/importance_sampling_ratio/mean": 0.9803248643875122, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3452838510274887, "entropy": 3.59776571393013, "clip_ratio/low_mean": 0.051804156973958015, "clip_ratio/low_min": 0.051804156973958015, "clip_ratio/high_mean": 0.1951428446918726, "clip_ratio/high_max": 0.1951428446918726, "clip_ratio/region_mean": 0.2469470016658306, "reward_total_mean": 1.1042048931121826, "reward_meter_mean": 0.9042049646377563, "reward_meter_std": 0.14156009256839752, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 41 |
+
{"timestamp_utc": "2026-04-11T15:50:48Z", "mode": "train", "global_step": 41, "epoch": 0.0015832561013283905, "loss": 0.1829, "grad_norm": 5.928605556488037, "learning_rate": 9.600000000000001e-06, "num_tokens": 83315.0, "completions/mean_length": 264.0, "completions/min_length": 215.0, "completions/max_length": 365.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 264.0, "completions/min_terminated_length": 215.0, "completions/max_terminated_length": 365.0, "rewards/meter/mean": 0.6348835229873657, "rewards/meter/std": 0.3371293842792511, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.784883439540863, "reward_std": 0.3861341178417206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15597331523895264, "sampling/sampling_logp_difference/max": 1.3521299362182617, "sampling/importance_sampling_ratio/min": 0.3288605809211731, "sampling/importance_sampling_ratio/mean": 0.9977389574050903, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0418922081589699, "entropy": 3.7767926454544067, "clip_ratio/low_mean": 0.06170275993645191, "clip_ratio/low_min": 0.06170275993645191, "clip_ratio/high_mean": 0.14350228570401669, "clip_ratio/high_max": 0.14350228570401669, "clip_ratio/region_mean": 0.2052050456404686, "reward_total_mean": 0.784883439540863, "reward_meter_mean": 0.6348835229873657, "reward_meter_std": 0.3371293842792511, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968}
|
| 42 |
+
{"timestamp_utc": "2026-04-11T15:50:54Z", "mode": "train", "global_step": 42, "epoch": 0.0016218721037998146, "loss": -0.0513, "grad_norm": 18.00514030456543, "learning_rate": 9.59e-06, "num_tokens": 85196.0, "completions/mean_length": 65.125, "completions/min_length": 48.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.43711739778518677, "rewards/meter/std": 0.3802741467952728, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6121174097061157, "reward_std": 0.3986560106277466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17133519053459167, "sampling/sampling_logp_difference/max": 1.1742663383483887, "sampling/importance_sampling_ratio/min": 0.3090456426143646, "sampling/importance_sampling_ratio/mean": 0.9917340874671936, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.2616077810525894, "entropy": 3.121804028749466, "clip_ratio/low_mean": 0.11971818003803492, "clip_ratio/low_min": 0.11971818003803492, "clip_ratio/high_mean": 0.08258083090186119, "clip_ratio/high_max": 0.08258083090186119, "clip_ratio/region_mean": 0.2022990109398961, "reward_total_mean": 0.6121174097061157, "reward_meter_mean": 0.43711739778518677, "reward_meter_std": 0.3802741467952728, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 43 |
+
{"timestamp_utc": "2026-04-11T15:50:59Z", "mode": "train", "global_step": 43, "epoch": 0.0016604881062712389, "loss": -0.0307, "grad_norm": 16.364044189453125, "learning_rate": 9.58e-06, "num_tokens": 86814.0, "completions/mean_length": 47.25, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.45132896304130554, "rewards/meter/std": 0.45054808259010315, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6513289213180542, "reward_std": 0.45054808259010315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19375590980052948, "sampling/sampling_logp_difference/max": 1.3071508407592773, "sampling/importance_sampling_ratio/min": 0.2705899178981781, "sampling/importance_sampling_ratio/mean": 0.996967077255249, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4401512742042542, "entropy": 2.3568301051855087, "clip_ratio/low_mean": 0.09967258013784885, "clip_ratio/low_min": 0.09967258013784885, "clip_ratio/high_mean": 0.1285256128758192, "clip_ratio/high_max": 0.1285256128758192, "clip_ratio/region_mean": 0.22819819301366806, "reward_total_mean": 0.6513289213180542, "reward_meter_mean": 0.45132896304130554, "reward_meter_std": 0.45054808259010315, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 44 |
+
{"timestamp_utc": "2026-04-11T15:51:04Z", "mode": "train", "global_step": 44, "epoch": 0.001699104108742663, "loss": 0.0607, "grad_norm": 23.743331909179688, "learning_rate": 9.57e-06, "num_tokens": 88539.0, "completions/mean_length": 43.625, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6140779256820679, "rewards/meter/std": 0.3976222574710846, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8140779733657837, "reward_std": 0.3976222574710846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1653001308441162, "sampling/sampling_logp_difference/max": 3.035217761993408, "sampling/importance_sampling_ratio/min": 0.04806419834494591, "sampling/importance_sampling_ratio/mean": 0.9870915412902832, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.507635399699211, "entropy": 0.9922648519277573, "clip_ratio/low_mean": 0.027404863387346268, "clip_ratio/low_min": 0.027404863387346268, "clip_ratio/high_mean": 0.10046137310564518, "clip_ratio/high_max": 0.10046137310564518, "clip_ratio/region_mean": 0.12786623649299145, "reward_total_mean": 0.8140779733657837, "reward_meter_mean": 0.6140779256820679, "reward_meter_std": 0.3976222574710846, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 45 |
+
{"timestamp_utc": "2026-04-11T15:51:08Z", "mode": "train", "global_step": 45, "epoch": 0.001737720111214087, "loss": 0.1034, "grad_norm": 17.9189510345459, "learning_rate": 9.56e-06, "num_tokens": 90315.0, "completions/mean_length": 38.0, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.433235764503479, "rewards/meter/std": 0.4090244174003601, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6332358121871948, "reward_std": 0.4090244770050049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19081313908100128, "sampling/sampling_logp_difference/max": 1.0551395416259766, "sampling/importance_sampling_ratio/min": 0.3481438159942627, "sampling/importance_sampling_ratio/mean": 0.9787829518318176, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.6995926201343536, "entropy": 2.8701943159103394, "clip_ratio/low_mean": 0.09415853396058083, "clip_ratio/low_min": 0.09415853396058083, "clip_ratio/high_mean": 0.09030752815306187, "clip_ratio/high_max": 0.09030752815306187, "clip_ratio/region_mean": 0.1844660621136427, "reward_total_mean": 0.6332358121871948, "reward_meter_mean": 0.433235764503479, "reward_meter_std": 0.4090244174003601, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 46 |
+
{"timestamp_utc": "2026-04-11T15:51:13Z", "mode": "train", "global_step": 46, "epoch": 0.0017763361136855112, "loss": 0.0292, "grad_norm": 35.99632263183594, "learning_rate": 9.55e-06, "num_tokens": 91656.0, "completions/mean_length": 29.625, "completions/min_length": 22.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.087132528424263, "rewards/meter/std": 0.08936810493469238, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.28713253140449524, "reward_std": 0.08936809748411179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2213541567325592, "sampling/sampling_logp_difference/max": 2.1127161979675293, "sampling/importance_sampling_ratio/min": 0.12090910226106644, "sampling/importance_sampling_ratio/mean": 0.9919323921203613, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7983057498931885, "entropy": 1.8424254208803177, "clip_ratio/low_mean": 0.14126449823379517, "clip_ratio/low_min": 0.14126449823379517, "clip_ratio/high_mean": 0.08075684309005737, "clip_ratio/high_max": 0.08075684309005737, "clip_ratio/region_mean": 0.22202134132385254, "reward_total_mean": 0.28713253140449524, "reward_meter_mean": 0.087132528424263, "reward_meter_std": 0.08936810493469238, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 47 |
+
{"timestamp_utc": "2026-04-11T15:51:17Z", "mode": "train", "global_step": 47, "epoch": 0.0018149521161569355, "loss": 0.0968, "grad_norm": 36.801151275634766, "learning_rate": 9.54e-06, "num_tokens": 93025.0, "completions/mean_length": 24.125, "completions/min_length": 24.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8150457739830017, "rewards/meter/std": 0.03531856834888458, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.0150457620620728, "reward_std": 0.03531854599714279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041083987802267075, "sampling/sampling_logp_difference/max": 1.2403922080993652, "sampling/importance_sampling_ratio/min": 0.2892707586288452, "sampling/importance_sampling_ratio/mean": 0.9999908208847046, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.1884380280971527, "entropy": 0.34495257679373026, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.0052083334885537624, "clip_ratio/high_max": 0.0052083334885537624, "clip_ratio/region_mean": 0.030208333861082792, "reward_total_mean": 1.0150457620620728, "reward_meter_mean": 0.8150457739830017, "reward_meter_std": 0.03531856834888458, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 48 |
+
{"timestamp_utc": "2026-04-11T15:51:23Z", "mode": "train", "global_step": 48, "epoch": 0.0018535681186283596, "loss": 0.0324, "grad_norm": 17.312498092651367, "learning_rate": 9.53e-06, "num_tokens": 95089.0, "completions/mean_length": 87.0, "completions/min_length": 52.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.23315757513046265, "rewards/meter/std": 0.06710986793041229, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4331575632095337, "reward_std": 0.06710987538099289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18739104270935059, "sampling/sampling_logp_difference/max": 1.2994401454925537, "sampling/importance_sampling_ratio/min": 0.2726844251155853, "sampling/importance_sampling_ratio/mean": 0.9660635590553284, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.2894017100334167, "entropy": 2.6435008347034454, "clip_ratio/low_mean": 0.10152440890669823, "clip_ratio/low_min": 0.10152440890669823, "clip_ratio/high_mean": 0.1315415445715189, "clip_ratio/high_max": 0.1315415445715189, "clip_ratio/region_mean": 0.23306595347821712, "reward_total_mean": 0.4331575632095337, "reward_meter_mean": 0.23315757513046265, "reward_meter_std": 0.06710986793041229, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 49 |
+
{"timestamp_utc": "2026-04-11T15:51:28Z", "mode": "train", "global_step": 49, "epoch": 0.0018921841210997837, "loss": 0.0945, "grad_norm": 15.068387031555176, "learning_rate": 9.52e-06, "num_tokens": 96846.0, "completions/mean_length": 43.625, "completions/min_length": 30.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.471576452255249, "rewards/meter/std": 0.4784705936908722, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6715764999389648, "reward_std": 0.4784706234931946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17449823021888733, "sampling/sampling_logp_difference/max": 1.1681387424468994, "sampling/importance_sampling_ratio/min": 0.3757398724555969, "sampling/importance_sampling_ratio/mean": 1.0054690837860107, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4131104350090027, "entropy": 2.8474780917167664, "clip_ratio/low_mean": 0.11420114897191525, "clip_ratio/low_min": 0.11420114897191525, "clip_ratio/high_mean": 0.11550478264689445, "clip_ratio/high_max": 0.11550478264689445, "clip_ratio/region_mean": 0.2297059316188097, "reward_total_mean": 0.6715764999389648, "reward_meter_mean": 0.471576452255249, "reward_meter_std": 0.4784705936908722, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 50 |
+
{"timestamp_utc": "2026-04-11T15:51:36Z", "mode": "train", "global_step": 50, "epoch": 0.0019308001235712078, "loss": 0.0575, "grad_norm": 8.942173957824707, "learning_rate": 9.51e-06, "num_tokens": 99306.0, "completions/mean_length": 130.5, "completions/min_length": 88.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.4820150136947632, "rewards/meter/std": 0.2984078824520111, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6570150256156921, "reward_std": 0.32944732904434204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.167738139629364, "sampling/sampling_logp_difference/max": 1.5406970977783203, "sampling/importance_sampling_ratio/min": 0.2918128967285156, "sampling/importance_sampling_ratio/mean": 0.9979844093322754, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2031668797135353, "entropy": 3.4669112861156464, "clip_ratio/low_mean": 0.07416268065571785, "clip_ratio/low_min": 0.07416268065571785, "clip_ratio/high_mean": 0.14113500528037548, "clip_ratio/high_max": 0.14113500528037548, "clip_ratio/region_mean": 0.21529768593609333, "reward_total_mean": 0.6570150256156921, "reward_meter_mean": 0.4820150136947632, "reward_meter_std": 0.2984078824520111, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652}
|
| 51 |
+
{"timestamp_utc": "2026-04-11T15:55:14Z", "mode": "eval", "global_step": 50, "epoch": 0.0019308001235712078, "eval_loss": 0.018872085958719254, "eval_runtime": 217.5742, "eval_samples_per_second": 0.478, "eval_steps_per_second": 0.06, "eval_num_tokens": 99306.0, "eval_completions/mean_length": 153.05769230769232, "eval_completions/min_length": 53.07692307692308, "eval_completions/max_length": 320.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 153.05769230769232, "eval_completions/min_terminated_length": 53.07692307692308, "eval_completions/max_terminated_length": 320.3076923076923, "eval_rewards/meter/mean": 0.2994799860394918, "eval_rewards/meter/std": 0.28621239931537557, "eval_rewards/exact_count_bonus/mean": 0.7307692307692307, "eval_rewards/exact_count_bonus/std": 0.3936365097761154, "eval_reward": 0.4456338297862273, "eval_reward_std": 0.1990389978656402, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03544207819952415, "eval_sampling/sampling_logp_difference/max": 0.29305177239271313, "eval_sampling/importance_sampling_ratio/min": 0.7560820785852579, "eval_sampling/importance_sampling_ratio/mean": 0.9996037987562326, "eval_sampling/importance_sampling_ratio/max": 1.2880034492566035, "eval_kl": 1.2056316297787886, "eval_entropy": 3.3071532799647403, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4456338297862273, "eval_reward_meter_mean": 0.2994799860394918, "eval_reward_meter_std": 0.28621239931537557, "eval_reward_exact_count_bonus_mean": 0.7307692307692307, "eval_reward_exact_count_bonus_std": 0.3936365097761154}
|
| 52 |
+
{"timestamp_utc": "2026-04-11T15:55:20Z", "mode": "train", "global_step": 51, "epoch": 0.001969416126042632, "loss": 0.1217, "grad_norm": 22.3005428314209, "learning_rate": 9.5e-06, "num_tokens": 100856.0, "completions/mean_length": 35.75, "completions/min_length": 27.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.2650821805000305, "rewards/meter/std": 0.4003763198852539, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.46508222818374634, "reward_std": 0.4003763198852539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2505285441875458, "sampling/sampling_logp_difference/max": 1.2456424236297607, "sampling/importance_sampling_ratio/min": 0.2877559959888458, "sampling/importance_sampling_ratio/mean": 0.9820157289505005, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8204183876514435, "entropy": 2.8815866708755493, "clip_ratio/low_mean": 0.15752424113452435, "clip_ratio/low_min": 0.15752424113452435, "clip_ratio/high_mean": 0.061693549156188965, "clip_ratio/high_max": 0.061693549156188965, "clip_ratio/region_mean": 0.2192177902907133, "reward_total_mean": 0.46508222818374634, "reward_meter_mean": 0.2650821805000305, "reward_meter_std": 0.4003763198852539, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
| 53 |
+
{"timestamp_utc": "2026-04-11T15:55:26Z", "mode": "train", "global_step": 52, "epoch": 0.002008032128514056, "loss": 0.0338, "grad_norm": 12.406262397766113, "learning_rate": 9.49e-06, "num_tokens": 102767.0, "completions/mean_length": 64.875, "completions/min_length": 40.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9629589319229126, "rewards/meter/std": 0.05814887955784798, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1629588603973389, "reward_std": 0.058148909360170364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13121938705444336, "sampling/sampling_logp_difference/max": 0.9225292205810547, "sampling/importance_sampling_ratio/min": 0.39751237630844116, "sampling/importance_sampling_ratio/mean": 1.004599928855896, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2337721958756447, "entropy": 2.761542499065399, "clip_ratio/low_mean": 0.013059701770544052, "clip_ratio/low_min": 0.013059701770544052, "clip_ratio/high_mean": 0.16483690962195396, "clip_ratio/high_max": 0.16483690962195396, "clip_ratio/region_mean": 0.17789661139249802, "reward_total_mean": 1.1629588603973389, "reward_meter_mean": 0.9629589319229126, "reward_meter_std": 0.05814887955784798, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0}
|
plots/chain_metrics.jsonl
CHANGED
|
@@ -1,5 +1,53 @@
|
|
| 1 |
-
{"timestamp_utc": "2026-04-
|
| 2 |
-
{"timestamp_utc": "2026-04-
|
| 3 |
-
{"timestamp_utc": "2026-04-
|
| 4 |
-
{"timestamp_utc": "2026-04-
|
| 5 |
-
{"timestamp_utc": "2026-04-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"timestamp_utc": "2026-04-11T15:46:43Z", "mode": "train", "global_step": 1, "epoch": 3.861600247142416e-05, "loss": 0.0875, "grad_norm": 14.433236122131348, "learning_rate": 1e-05, "num_tokens": 2073.0, "completions/mean_length": 72.125, "completions/min_length": 68.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 72.125, "completions/min_terminated_length": 68.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.7467405796051025, "rewards/meter/std": 0.2669501304626465, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9467406272888184, "reward_std": 0.26695016026496887, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1810092180967331, "sampling/sampling_logp_difference/max": 2.3954362869262695, "sampling/importance_sampling_ratio/min": 0.09113290905952454, "sampling/importance_sampling_ratio/mean": 0.9819375872612, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.716715857386589, "entropy": 2.8896071165800095, "clip_ratio/low_mean": 0.09221794456243515, "clip_ratio/low_min": 0.09221794456243515, "clip_ratio/high_mean": 0.16360880248248577, "clip_ratio/high_max": 0.16360880248248577, "clip_ratio/region_mean": 0.2558267470449209, "reward_total_mean": 0.9467406272888184, "reward_meter_mean": 0.7467405796051025, "reward_meter_std": 0.2669501304626465, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 2 |
+
{"timestamp_utc": "2026-04-11T15:46:49Z", "mode": "train", "global_step": 2, "epoch": 7.723200494284832e-05, "loss": 0.1013, "grad_norm": 14.803139686584473, "learning_rate": 9.990000000000001e-06, "num_tokens": 3899.0, "completions/mean_length": 59.25, "completions/min_length": 44.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.25, "completions/min_terminated_length": 44.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.692160427570343, "rewards/meter/std": 0.3035436272621155, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8921604156494141, "reward_std": 0.3035435974597931, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17282956838607788, "sampling/sampling_logp_difference/max": 1.6599078178405762, "sampling/importance_sampling_ratio/min": 0.27146539092063904, "sampling/importance_sampling_ratio/mean": 0.9748801589012146, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5766338407993317, "entropy": 1.2240338400006294, "clip_ratio/low_mean": 0.037109375, "clip_ratio/low_min": 0.037109375, "clip_ratio/high_mean": 0.15457439702004194, "clip_ratio/high_max": 0.15457439702004194, "clip_ratio/region_mean": 0.19168377202004194, "reward_total_mean": 0.8921604156494141, "reward_meter_mean": 0.692160427570343, "reward_meter_std": 0.3035436272621155, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 3 |
+
{"timestamp_utc": "2026-04-11T15:46:53Z", "mode": "train", "global_step": 3, "epoch": 0.00011584800741427248, "loss": 0.1019, "grad_norm": 22.877466201782227, "learning_rate": 9.980000000000001e-06, "num_tokens": 5552.0, "completions/mean_length": 40.625, "completions/min_length": 32.0, "completions/max_length": 51.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 40.625, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 51.0, "rewards/meter/mean": 0.5208516120910645, "rewards/meter/std": 0.4123581349849701, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7208516597747803, "reward_std": 0.4123581051826477, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18985480070114136, "sampling/sampling_logp_difference/max": 1.3331613540649414, "sampling/importance_sampling_ratio/min": 0.26364248991012573, "sampling/importance_sampling_ratio/mean": 0.9614238142967224, "sampling/importance_sampling_ratio/max": 1.952972412109375, "kl": 2.4038615822792053, "entropy": 2.205263152718544, "clip_ratio/low_mean": 0.08594596944749355, "clip_ratio/low_min": 0.08594596944749355, "clip_ratio/high_mean": 0.11026187054812908, "clip_ratio/high_max": 0.11026187054812908, "clip_ratio/region_mean": 0.19620783999562263, "reward_total_mean": 0.7208516597747803, "reward_meter_mean": 0.5208516120910645, "reward_meter_std": 0.4123581349849701, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 4 |
+
{"timestamp_utc": "2026-04-11T15:46:58Z", "mode": "train", "global_step": 4, "epoch": 0.00015446400988569664, "loss": 0.0914, "grad_norm": 23.66639518737793, "learning_rate": 9.970000000000001e-06, "num_tokens": 6962.0, "completions/mean_length": 22.25, "completions/min_length": 19.0, "completions/max_length": 26.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.25, "completions/min_terminated_length": 19.0, "completions/max_terminated_length": 26.0, "rewards/meter/mean": 0.7492154836654663, "rewards/meter/std": 0.4238397777080536, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9492155313491821, "reward_std": 0.4238397479057312, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.11156373471021652, "sampling/sampling_logp_difference/max": 1.2559912204742432, "sampling/importance_sampling_ratio/min": 0.507404625415802, "sampling/importance_sampling_ratio/mean": 1.0157339572906494, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5658123940229416, "entropy": 1.1889886930584908, "clip_ratio/low_mean": 0.04130434803664684, "clip_ratio/low_min": 0.04130434803664684, "clip_ratio/high_mean": 0.09267808869481087, "clip_ratio/high_max": 0.09267808869481087, "clip_ratio/region_mean": 0.1339824367314577, "reward_total_mean": 0.9492155313491821, "reward_meter_mean": 0.7492154836654663, "reward_meter_std": 0.4238397777080536, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 5 |
+
{"timestamp_utc": "2026-04-11T15:47:07Z", "mode": "train", "global_step": 5, "epoch": 0.0001930800123571208, "loss": -0.0187, "grad_norm": 9.979012489318848, "learning_rate": 9.960000000000001e-06, "num_tokens": 9821.0, "completions/mean_length": 152.375, "completions/min_length": 99.0, "completions/max_length": 235.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 152.375, "completions/min_terminated_length": 99.0, "completions/max_terminated_length": 235.0, "rewards/meter/mean": 0.31202924251556396, "rewards/meter/std": 0.2894730269908905, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5120292901992798, "reward_std": 0.2894730269908905, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19261623919010162, "sampling/sampling_logp_difference/max": 1.489715576171875, "sampling/importance_sampling_ratio/min": 0.340789794921875, "sampling/importance_sampling_ratio/mean": 0.987079381942749, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.233508050441742, "entropy": 3.3276279270648956, "clip_ratio/low_mean": 0.14424454979598522, "clip_ratio/low_min": 0.14424454979598522, "clip_ratio/high_mean": 0.07602224312722683, "clip_ratio/high_max": 0.07602224312722683, "clip_ratio/region_mean": 0.22026679292321205, "reward_total_mean": 0.5120292901992798, "reward_meter_mean": 0.31202924251556396, "reward_meter_std": 0.2894730269908905, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 6 |
+
{"timestamp_utc": "2026-04-11T15:47:12Z", "mode": "train", "global_step": 6, "epoch": 0.00023169601482854495, "loss": 0.1184, "grad_norm": 15.483458518981934, "learning_rate": 9.950000000000001e-06, "num_tokens": 11697.0, "completions/mean_length": 76.5, "completions/min_length": 61.0, "completions/max_length": 100.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 76.5, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 100.0, "rewards/meter/mean": 0.7063730955123901, "rewards/meter/std": 0.21755146980285645, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9063730835914612, "reward_std": 0.21755146980285645, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21761518716812134, "sampling/sampling_logp_difference/max": 1.3374531269073486, "sampling/importance_sampling_ratio/min": 0.2625133991241455, "sampling/importance_sampling_ratio/mean": 0.9750673770904541, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.003227472305298, "entropy": 3.393460273742676, "clip_ratio/low_mean": 0.09787485934793949, "clip_ratio/low_min": 0.09787485934793949, "clip_ratio/high_mean": 0.1323312446475029, "clip_ratio/high_max": 0.1323312446475029, "clip_ratio/region_mean": 0.2302061039954424, "reward_total_mean": 0.9063730835914612, "reward_meter_mean": 0.7063730955123901, "reward_meter_std": 0.21755146980285645, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 7 |
+
{"timestamp_utc": "2026-04-11T15:47:19Z", "mode": "train", "global_step": 7, "epoch": 0.0002703120172999691, "loss": 0.2978, "grad_norm": 16.140308380126953, "learning_rate": 9.940000000000001e-06, "num_tokens": 13816.0, "completions/mean_length": 93.875, "completions/min_length": 62.0, "completions/max_length": 135.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 93.875, "completions/min_terminated_length": 62.0, "completions/max_terminated_length": 135.0, "rewards/meter/mean": 0.4120001196861267, "rewards/meter/std": 0.33185702562332153, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.5870001316070557, "reward_std": 0.36883509159088135, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21371272206306458, "sampling/sampling_logp_difference/max": 1.7315394878387451, "sampling/importance_sampling_ratio/min": 0.18673592805862427, "sampling/importance_sampling_ratio/mean": 0.9866839051246643, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.0300692468881607, "entropy": 2.9779615700244904, "clip_ratio/low_mean": 0.12269405275583267, "clip_ratio/low_min": 0.12269405275583267, "clip_ratio/high_mean": 0.07538295164704323, "clip_ratio/high_max": 0.07538295164704323, "clip_ratio/region_mean": 0.1980770044028759, "reward_total_mean": 0.5870001316070557, "reward_meter_mean": 0.4120001196861267, "reward_meter_std": 0.33185702562332153, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 8 |
+
{"timestamp_utc": "2026-04-11T15:47:24Z", "mode": "train", "global_step": 8, "epoch": 0.00030892801977139327, "loss": 0.0345, "grad_norm": 20.1074161529541, "learning_rate": 9.930000000000001e-06, "num_tokens": 15480.0, "completions/mean_length": 42.0, "completions/min_length": 28.0, "completions/max_length": 54.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.0, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 54.0, "rewards/meter/mean": 0.5687531232833862, "rewards/meter/std": 0.3685603737831116, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.7437530755996704, "reward_std": 0.39819100499153137, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19006481766700745, "sampling/sampling_logp_difference/max": 1.9432754516601562, "sampling/importance_sampling_ratio/min": 0.1432340294122696, "sampling/importance_sampling_ratio/mean": 0.9912540316581726, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.0594891011714935, "entropy": 2.9420621395111084, "clip_ratio/low_mean": 0.09676836617290974, "clip_ratio/low_min": 0.09676836617290974, "clip_ratio/high_mean": 0.10601851902902126, "clip_ratio/high_max": 0.10601851902902126, "clip_ratio/region_mean": 0.202786885201931, "reward_total_mean": 0.7437530755996704, "reward_meter_mean": 0.5687531232833862, "reward_meter_std": 0.3685603737831116, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 9 |
+
{"timestamp_utc": "2026-04-11T15:47:29Z", "mode": "train", "global_step": 9, "epoch": 0.00034754402224281743, "loss": 0.1313, "grad_norm": 21.095970153808594, "learning_rate": 9.920000000000002e-06, "num_tokens": 17114.0, "completions/mean_length": 47.25, "completions/min_length": 37.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 37.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.5934403538703918, "rewards/meter/std": 0.4387426972389221, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7934404015541077, "reward_std": 0.4387427270412445, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18523243069648743, "sampling/sampling_logp_difference/max": 2.2344932556152344, "sampling/importance_sampling_ratio/min": 0.10704635828733444, "sampling/importance_sampling_ratio/mean": 0.9886912107467651, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5875789374113083, "entropy": 2.3184936344623566, "clip_ratio/low_mean": 0.0800175815820694, "clip_ratio/low_min": 0.0800175815820694, "clip_ratio/high_mean": 0.14036413840949535, "clip_ratio/high_max": 0.14036413840949535, "clip_ratio/region_mean": 0.22038171999156475, "reward_total_mean": 0.7934404015541077, "reward_meter_mean": 0.5934403538703918, "reward_meter_std": 0.4387426972389221, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 10 |
+
{"timestamp_utc": "2026-04-11T15:47:34Z", "mode": "train", "global_step": 10, "epoch": 0.0003861600247142416, "loss": -0.0138, "grad_norm": 17.000186920166016, "learning_rate": 9.91e-06, "num_tokens": 18851.0, "completions/mean_length": 48.125, "completions/min_length": 33.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 48.125, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.4887843132019043, "rewards/meter/std": 0.42971470952033997, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6637843251228333, "reward_std": 0.45756226778030396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2051706463098526, "sampling/sampling_logp_difference/max": 1.3021516799926758, "sampling/importance_sampling_ratio/min": 0.3167512118816376, "sampling/importance_sampling_ratio/mean": 0.9984660148620605, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8380893766880035, "entropy": 3.0384787023067474, "clip_ratio/low_mean": 0.16380306333303452, "clip_ratio/low_min": 0.16380306333303452, "clip_ratio/high_mean": 0.09042712114751339, "clip_ratio/high_max": 0.09042712114751339, "clip_ratio/region_mean": 0.2542301844805479, "reward_total_mean": 0.6637843251228333, "reward_meter_mean": 0.4887843132019043, "reward_meter_std": 0.42971470952033997, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 11 |
+
{"timestamp_utc": "2026-04-11T15:47:39Z", "mode": "train", "global_step": 11, "epoch": 0.00042477602718566575, "loss": 0.1831, "grad_norm": 16.804243087768555, "learning_rate": 9.9e-06, "num_tokens": 20578.0, "completions/mean_length": 44.875, "completions/min_length": 29.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.30691099166870117, "rewards/meter/std": 0.4142797291278839, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.506911039352417, "reward_std": 0.4142796993255615, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18619823455810547, "sampling/sampling_logp_difference/max": 1.9143197536468506, "sampling/importance_sampling_ratio/min": 0.14744210243225098, "sampling/importance_sampling_ratio/mean": 0.9988991618156433, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.717022493481636, "entropy": 2.5153299272060394, "clip_ratio/low_mean": 0.11223171092569828, "clip_ratio/low_min": 0.11223171092569828, "clip_ratio/high_mean": 0.08737440872937441, "clip_ratio/high_max": 0.08737440872937441, "clip_ratio/region_mean": 0.1996061196550727, "reward_total_mean": 0.506911039352417, "reward_meter_mean": 0.30691099166870117, "reward_meter_std": 0.4142797291278839, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 12 |
+
{"timestamp_utc": "2026-04-11T15:47:44Z", "mode": "train", "global_step": 12, "epoch": 0.0004633920296570899, "loss": -0.02, "grad_norm": 22.950233459472656, "learning_rate": 9.89e-06, "num_tokens": 22530.0, "completions/mean_length": 58.0, "completions/min_length": 38.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 58.0, "completions/min_terminated_length": 38.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.2489539384841919, "rewards/meter/std": 0.23509462177753448, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4489539563655853, "reward_std": 0.23509462177753448, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2930803596973419, "sampling/sampling_logp_difference/max": 3.918379306793213, "sampling/importance_sampling_ratio/min": 0.07527641206979752, "sampling/importance_sampling_ratio/mean": 0.9429361820220947, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.302564397454262, "entropy": 1.546494573354721, "clip_ratio/low_mean": 0.13612079434096813, "clip_ratio/low_min": 0.13612079434096813, "clip_ratio/high_mean": 0.048309178091585636, "clip_ratio/high_max": 0.048309178091585636, "clip_ratio/region_mean": 0.18442997243255377, "reward_total_mean": 0.4489539563655853, "reward_meter_mean": 0.2489539384841919, "reward_meter_std": 0.23509462177753448, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 13 |
+
{"timestamp_utc": "2026-04-11T15:47:49Z", "mode": "train", "global_step": 13, "epoch": 0.000502008032128514, "loss": 0.0551, "grad_norm": 20.975563049316406, "learning_rate": 9.88e-06, "num_tokens": 24249.0, "completions/mean_length": 46.875, "completions/min_length": 33.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 46.875, "completions/min_terminated_length": 33.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.5702009201049805, "rewards/meter/std": 0.416922926902771, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7702009677886963, "reward_std": 0.41692298650741577, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19165004789829254, "sampling/sampling_logp_difference/max": 1.6037216186523438, "sampling/importance_sampling_ratio/min": 0.20114654302597046, "sampling/importance_sampling_ratio/mean": 0.9821399450302124, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.4146449714899063, "entropy": 2.7278781831264496, "clip_ratio/low_mean": 0.0789086427539587, "clip_ratio/low_min": 0.0789086427539587, "clip_ratio/high_mean": 0.11046412773430347, "clip_ratio/high_max": 0.11046412773430347, "clip_ratio/region_mean": 0.18937277048826218, "reward_total_mean": 0.7702009677886963, "reward_meter_mean": 0.5702009201049805, "reward_meter_std": 0.416922926902771, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 14 |
+
{"timestamp_utc": "2026-04-11T15:47:56Z", "mode": "train", "global_step": 14, "epoch": 0.0005406240345999382, "loss": 0.0636, "grad_norm": 13.641975402832031, "learning_rate": 9.87e-06, "num_tokens": 26453.0, "completions/mean_length": 82.5, "completions/min_length": 63.0, "completions/max_length": 126.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 82.5, "completions/min_terminated_length": 63.0, "completions/max_terminated_length": 126.0, "rewards/meter/mean": 0.3489101529121399, "rewards/meter/std": 0.34769579768180847, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5489101409912109, "reward_std": 0.34769579768180847, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2145778238773346, "sampling/sampling_logp_difference/max": 2.2805795669555664, "sampling/importance_sampling_ratio/min": 0.1022249385714531, "sampling/importance_sampling_ratio/mean": 0.9998252987861633, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3619689494371414, "entropy": 2.838075205683708, "clip_ratio/low_mean": 0.13307647220790386, "clip_ratio/low_min": 0.13307647220790386, "clip_ratio/high_mean": 0.09134172648191452, "clip_ratio/high_max": 0.09134172648191452, "clip_ratio/region_mean": 0.22441819868981838, "reward_total_mean": 0.5489101409912109, "reward_meter_mean": 0.3489101529121399, "reward_meter_std": 0.34769579768180847, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 15 |
+
{"timestamp_utc": "2026-04-11T15:48:00Z", "mode": "train", "global_step": 15, "epoch": 0.0005792400370713623, "loss": 0.1104, "grad_norm": 26.442886352539062, "learning_rate": 9.86e-06, "num_tokens": 27811.0, "completions/mean_length": 22.75, "completions/min_length": 15.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 22.75, "completions/min_terminated_length": 15.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.6123703718185425, "rewards/meter/std": 0.42484015226364136, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8123704195022583, "reward_std": 0.42484015226364136, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.21215009689331055, "sampling/sampling_logp_difference/max": 2.0358657836914062, "sampling/importance_sampling_ratio/min": 0.3356194794178009, "sampling/importance_sampling_ratio/mean": 0.9939978122711182, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7645654305815697, "entropy": 2.376413881778717, "clip_ratio/low_mean": 0.05330882500857115, "clip_ratio/low_min": 0.05330882500857115, "clip_ratio/high_mean": 0.11765297874808311, "clip_ratio/high_max": 0.11765297874808311, "clip_ratio/region_mean": 0.17096180375665426, "reward_total_mean": 0.8123704195022583, "reward_meter_mean": 0.6123703718185425, "reward_meter_std": 0.42484015226364136, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 16 |
+
{"timestamp_utc": "2026-04-11T15:48:05Z", "mode": "train", "global_step": 16, "epoch": 0.0006178560395427865, "loss": 0.105, "grad_norm": 20.25496482849121, "learning_rate": 9.85e-06, "num_tokens": 29402.0, "completions/mean_length": 34.875, "completions/min_length": 29.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 34.875, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.59489905834198, "rewards/meter/std": 0.4257949888706207, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7948991060256958, "reward_std": 0.42579495906829834, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2184409201145172, "sampling/sampling_logp_difference/max": 1.4742670059204102, "sampling/importance_sampling_ratio/min": 0.24731610715389252, "sampling/importance_sampling_ratio/mean": 0.967242956161499, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.90836800634861, "entropy": 2.617787539958954, "clip_ratio/low_mean": 0.10445018857717514, "clip_ratio/low_min": 0.10445018857717514, "clip_ratio/high_mean": 0.10188646428287029, "clip_ratio/high_max": 0.10188646428287029, "clip_ratio/region_mean": 0.20633665286004543, "reward_total_mean": 0.7948991060256958, "reward_meter_mean": 0.59489905834198, "reward_meter_std": 0.4257949888706207, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 17 |
+
{"timestamp_utc": "2026-04-11T15:48:09Z", "mode": "train", "global_step": 17, "epoch": 0.0006564720420142106, "loss": 0.1285, "grad_norm": 25.74098014831543, "learning_rate": 9.84e-06, "num_tokens": 30963.0, "completions/mean_length": 41.125, "completions/min_length": 26.0, "completions/max_length": 49.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.125, "completions/min_terminated_length": 26.0, "completions/max_terminated_length": 49.0, "rewards/meter/mean": 0.8503542542457581, "rewards/meter/std": 0.3227464556694031, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.050354242324829, "reward_std": 0.32274648547172546, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.10933797806501389, "sampling/sampling_logp_difference/max": 1.4986724853515625, "sampling/importance_sampling_ratio/min": 0.22342658042907715, "sampling/importance_sampling_ratio/mean": 0.9962012767791748, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7420838475227356, "entropy": 0.8714146893471479, "clip_ratio/low_mean": 0.059752749279141426, "clip_ratio/low_min": 0.059752749279141426, "clip_ratio/high_mean": 0.056653511710464954, "clip_ratio/high_max": 0.056653511710464954, "clip_ratio/region_mean": 0.11640626098960638, "reward_total_mean": 1.050354242324829, "reward_meter_mean": 0.8503542542457581, "reward_meter_std": 0.3227464556694031, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 18 |
+
{"timestamp_utc": "2026-04-11T15:48:14Z", "mode": "train", "global_step": 18, "epoch": 0.0006950880444856349, "loss": 0.1207, "grad_norm": 17.443212509155273, "learning_rate": 9.83e-06, "num_tokens": 32520.0, "completions/mean_length": 42.625, "completions/min_length": 31.0, "completions/max_length": 61.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 42.625, "completions/min_terminated_length": 31.0, "completions/max_terminated_length": 61.0, "rewards/meter/mean": 0.31625181436538696, "rewards/meter/std": 0.2878619432449341, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5162518620491028, "reward_std": 0.2878619432449341, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1679675132036209, "sampling/sampling_logp_difference/max": 0.9356918334960938, "sampling/importance_sampling_ratio/min": 0.3923143446445465, "sampling/importance_sampling_ratio/mean": 0.9975839257240295, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4733051806688309, "entropy": 2.5996862947940826, "clip_ratio/low_mean": 0.11914125084877014, "clip_ratio/low_min": 0.11914125084877014, "clip_ratio/high_mean": 0.07529962994158268, "clip_ratio/high_max": 0.07529962994158268, "clip_ratio/region_mean": 0.19444088079035282, "reward_total_mean": 0.5162518620491028, "reward_meter_mean": 0.31625181436538696, "reward_meter_std": 0.2878619432449341, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 19 |
+
{"timestamp_utc": "2026-04-11T15:48:22Z", "mode": "train", "global_step": 19, "epoch": 0.000733704046957059, "loss": 0.1359, "grad_norm": 8.90630054473877, "learning_rate": 9.820000000000001e-06, "num_tokens": 34880.0, "completions/mean_length": 119.0, "completions/min_length": 67.0, "completions/max_length": 169.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 119.0, "completions/min_terminated_length": 67.0, "completions/max_terminated_length": 169.0, "rewards/meter/mean": 0.4886695444583893, "rewards/meter/std": 0.21834495663642883, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.6386695504188538, "reward_std": 0.25513508915901184, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16941869258880615, "sampling/sampling_logp_difference/max": 1.298689365386963, "sampling/importance_sampling_ratio/min": 0.27288922667503357, "sampling/importance_sampling_ratio/mean": 1.001911997795105, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0056203603744507, "entropy": 2.8221229016780853, "clip_ratio/low_mean": 0.11502809636294842, "clip_ratio/low_min": 0.11502809636294842, "clip_ratio/high_mean": 0.08858394995331764, "clip_ratio/high_max": 0.08858394995331764, "clip_ratio/region_mean": 0.20361204631626606, "reward_total_mean": 0.6386695504188538, "reward_meter_mean": 0.4886695444583893, "reward_meter_std": 0.21834495663642883, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 20 |
+
{"timestamp_utc": "2026-04-11T15:48:27Z", "mode": "train", "global_step": 20, "epoch": 0.0007723200494284832, "loss": 0.1452, "grad_norm": 14.272367477416992, "learning_rate": 9.810000000000001e-06, "num_tokens": 36697.0, "completions/mean_length": 53.125, "completions/min_length": 35.0, "completions/max_length": 73.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 53.125, "completions/min_terminated_length": 35.0, "completions/max_terminated_length": 73.0, "rewards/meter/mean": 0.584618330001831, "rewards/meter/std": 0.37843555212020874, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7846182584762573, "reward_std": 0.37843549251556396, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18555279076099396, "sampling/sampling_logp_difference/max": 1.712082862854004, "sampling/importance_sampling_ratio/min": 0.1804894655942917, "sampling/importance_sampling_ratio/mean": 0.9827072620391846, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.500285528600216, "entropy": 2.9686510264873505, "clip_ratio/low_mean": 0.1128854975104332, "clip_ratio/low_min": 0.1128854975104332, "clip_ratio/high_mean": 0.12786824442446232, "clip_ratio/high_max": 0.12786824442446232, "clip_ratio/region_mean": 0.24075374193489552, "reward_total_mean": 0.7846182584762573, "reward_meter_mean": 0.584618330001831, "reward_meter_std": 0.37843555212020874, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 21 |
+
{"timestamp_utc": "2026-04-11T15:48:34Z", "mode": "train", "global_step": 21, "epoch": 0.0008109360518999073, "loss": 0.0602, "grad_norm": 8.9111328125, "learning_rate": 9.800000000000001e-06, "num_tokens": 39216.0, "completions/mean_length": 127.875, "completions/min_length": 88.0, "completions/max_length": 163.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 127.875, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 163.0, "rewards/meter/mean": 0.5299741625785828, "rewards/meter/std": 0.38676267862319946, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7299741506576538, "reward_std": 0.38676267862319946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1731364130973816, "sampling/sampling_logp_difference/max": 1.4290367364883423, "sampling/importance_sampling_ratio/min": 0.23953956365585327, "sampling/importance_sampling_ratio/mean": 0.9976959228515625, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3634046390652657, "entropy": 3.330483376979828, "clip_ratio/low_mean": 0.1339168418198824, "clip_ratio/low_min": 0.1339168418198824, "clip_ratio/high_mean": 0.08470636792480946, "clip_ratio/high_max": 0.08470636792480946, "clip_ratio/region_mean": 0.21862320974469185, "reward_total_mean": 0.7299741506576538, "reward_meter_mean": 0.5299741625785828, "reward_meter_std": 0.38676267862319946, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 22 |
+
{"timestamp_utc": "2026-04-11T15:48:39Z", "mode": "train", "global_step": 22, "epoch": 0.0008495520543713315, "loss": 0.0956, "grad_norm": 19.450031280517578, "learning_rate": 9.790000000000001e-06, "num_tokens": 40839.0, "completions/mean_length": 41.875, "completions/min_length": 32.0, "completions/max_length": 50.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 41.875, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 50.0, "rewards/meter/mean": 0.7068014740943909, "rewards/meter/std": 0.39332520961761475, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9068014621734619, "reward_std": 0.39332520961761475, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19435128569602966, "sampling/sampling_logp_difference/max": 1.2685363292694092, "sampling/importance_sampling_ratio/min": 0.2812429666519165, "sampling/importance_sampling_ratio/mean": 0.9838151931762695, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.744483582675457, "entropy": 1.974789410829544, "clip_ratio/low_mean": 0.09037221781909466, "clip_ratio/low_min": 0.09037221781909466, "clip_ratio/high_mean": 0.11499525140970945, "clip_ratio/high_max": 0.11499525140970945, "clip_ratio/region_mean": 0.2053674692288041, "reward_total_mean": 0.9068014621734619, "reward_meter_mean": 0.7068014740943909, "reward_meter_std": 0.39332520961761475, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 23 |
+
{"timestamp_utc": "2026-04-11T15:48:44Z", "mode": "train", "global_step": 23, "epoch": 0.0008881680568427556, "loss": -0.0134, "grad_norm": 11.986440658569336, "learning_rate": 9.780000000000001e-06, "num_tokens": 42619.0, "completions/mean_length": 59.5, "completions/min_length": 51.0, "completions/max_length": 72.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 59.5, "completions/min_terminated_length": 51.0, "completions/max_terminated_length": 72.0, "rewards/meter/mean": 0.3556836247444153, "rewards/meter/std": 0.41377386450767517, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5556836128234863, "reward_std": 0.4137738347053528, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15584221482276917, "sampling/sampling_logp_difference/max": 0.9905836582183838, "sampling/importance_sampling_ratio/min": 0.37629470229148865, "sampling/importance_sampling_ratio/mean": 0.9997124671936035, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2204247117042542, "entropy": 2.614344671368599, "clip_ratio/low_mean": 0.12430914491415024, "clip_ratio/low_min": 0.12430914491415024, "clip_ratio/high_mean": 0.0615717563778162, "clip_ratio/high_max": 0.0615717563778162, "clip_ratio/region_mean": 0.18588090129196644, "reward_total_mean": 0.5556836128234863, "reward_meter_mean": 0.3556836247444153, "reward_meter_std": 0.41377386450767517, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 24 |
+
{"timestamp_utc": "2026-04-11T15:48:49Z", "mode": "train", "global_step": 24, "epoch": 0.0009267840593141798, "loss": 0.0328, "grad_norm": 15.669774055480957, "learning_rate": 9.770000000000001e-06, "num_tokens": 44409.0, "completions/mean_length": 54.75, "completions/min_length": 29.0, "completions/max_length": 68.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 54.75, "completions/min_terminated_length": 29.0, "completions/max_terminated_length": 68.0, "rewards/meter/mean": 0.5161629915237427, "rewards/meter/std": 0.41966450214385986, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6911630034446716, "reward_std": 0.40840157866477966, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1744653582572937, "sampling/sampling_logp_difference/max": 1.210925817489624, "sampling/importance_sampling_ratio/min": 0.2979213297367096, "sampling/importance_sampling_ratio/mean": 1.0036118030548096, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.9536623805761337, "entropy": 2.8385555744171143, "clip_ratio/low_mean": 0.10080079734325409, "clip_ratio/low_min": 0.10080079734325409, "clip_ratio/high_mean": 0.08839947171509266, "clip_ratio/high_max": 0.08839947171509266, "clip_ratio/region_mean": 0.18920026905834675, "reward_total_mean": 0.6911630034446716, "reward_meter_mean": 0.5161629915237427, "reward_meter_std": 0.41966450214385986, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 25 |
+
{"timestamp_utc": "2026-04-11T15:48:54Z", "mode": "train", "global_step": 25, "epoch": 0.0009654000617856039, "loss": 0.056, "grad_norm": 13.490758895874023, "learning_rate": 9.760000000000001e-06, "num_tokens": 46042.0, "completions/mean_length": 36.125, "completions/min_length": 25.0, "completions/max_length": 56.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 36.125, "completions/min_terminated_length": 25.0, "completions/max_terminated_length": 56.0, "rewards/meter/mean": 0.2736630439758301, "rewards/meter/std": 0.30189424753189087, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4736630320549011, "reward_std": 0.3018941879272461, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13724787533283234, "sampling/sampling_logp_difference/max": 1.1764202117919922, "sampling/importance_sampling_ratio/min": 0.3343762159347534, "sampling/importance_sampling_ratio/mean": 0.9891919493675232, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5680749863386154, "entropy": 1.6488263756036758, "clip_ratio/low_mean": 0.12277928367257118, "clip_ratio/low_min": 0.12277928367257118, "clip_ratio/high_mean": 0.035714286379516125, "clip_ratio/high_max": 0.035714286379516125, "clip_ratio/region_mean": 0.1584935700520873, "reward_total_mean": 0.4736630320549011, "reward_meter_mean": 0.2736630439758301, "reward_meter_std": 0.30189424753189087, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 26 |
+
{"timestamp_utc": "2026-04-11T15:49:00Z", "mode": "train", "global_step": 26, "epoch": 0.001004016064257028, "loss": 0.202, "grad_norm": 16.0242919921875, "learning_rate": 9.75e-06, "num_tokens": 47681.0, "completions/mean_length": 44.875, "completions/min_length": 28.0, "completions/max_length": 67.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 28.0, "completions/max_terminated_length": 67.0, "rewards/meter/mean": 0.42073720693588257, "rewards/meter/std": 0.3384658396244049, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6207371950149536, "reward_std": 0.3384658098220825, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17049100995063782, "sampling/sampling_logp_difference/max": 1.2931162118911743, "sampling/importance_sampling_ratio/min": 0.2744143009185791, "sampling/importance_sampling_ratio/mean": 0.9966971278190613, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.6573083698749542, "entropy": 2.3349276185035706, "clip_ratio/low_mean": 0.10312438476830721, "clip_ratio/low_min": 0.10312438476830721, "clip_ratio/high_mean": 0.07951899617910385, "clip_ratio/high_max": 0.07951899617910385, "clip_ratio/region_mean": 0.18264338094741106, "reward_total_mean": 0.6207371950149536, "reward_meter_mean": 0.42073720693588257, "reward_meter_std": 0.3384658396244049, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 27 |
+
{"timestamp_utc": "2026-04-11T15:49:07Z", "mode": "train", "global_step": 27, "epoch": 0.0010426320667284523, "loss": 0.1939, "grad_norm": 21.260799407958984, "learning_rate": 9.74e-06, "num_tokens": 50118.0, "completions/mean_length": 114.625, "completions/min_length": 79.0, "completions/max_length": 157.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 114.625, "completions/min_terminated_length": 79.0, "completions/max_terminated_length": 157.0, "rewards/meter/mean": 0.42070716619491577, "rewards/meter/std": 0.32767125964164734, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.5957071781158447, "reward_std": 0.3487240672111511, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19498828053474426, "sampling/sampling_logp_difference/max": 1.8702855110168457, "sampling/importance_sampling_ratio/min": 0.15407966077327728, "sampling/importance_sampling_ratio/mean": 0.9976061582565308, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.746354952454567, "entropy": 2.8878106623888016, "clip_ratio/low_mean": 0.11959273181855679, "clip_ratio/low_min": 0.11959273181855679, "clip_ratio/high_mean": 0.11580519005656242, "clip_ratio/high_max": 0.11580519005656242, "clip_ratio/region_mean": 0.2353979218751192, "reward_total_mean": 0.5957071781158447, "reward_meter_mean": 0.42070716619491577, "reward_meter_std": 0.32767125964164734, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 28 |
+
{"timestamp_utc": "2026-04-11T15:49:13Z", "mode": "train", "global_step": 28, "epoch": 0.0010812480691998764, "loss": 0.1229, "grad_norm": 25.492298126220703, "learning_rate": 9.73e-06, "num_tokens": 52550.0, "completions/mean_length": 102.0, "completions/min_length": 86.0, "completions/max_length": 117.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 102.0, "completions/min_terminated_length": 86.0, "completions/max_terminated_length": 117.0, "rewards/meter/mean": 0.4947388172149658, "rewards/meter/std": 0.2714572250843048, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6697388291358948, "reward_std": 0.23900270462036133, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2764182388782501, "sampling/sampling_logp_difference/max": 2.778400421142578, "sampling/importance_sampling_ratio/min": 0.062137819826602936, "sampling/importance_sampling_ratio/mean": 0.9748236536979675, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.430370479822159, "entropy": 1.6325490325689316, "clip_ratio/low_mean": 0.12361728027462959, "clip_ratio/low_min": 0.12361728027462959, "clip_ratio/high_mean": 0.0886442456394434, "clip_ratio/high_max": 0.0886442456394434, "clip_ratio/region_mean": 0.212261525914073, "reward_total_mean": 0.6697388291358948, "reward_meter_mean": 0.4947388172149658, "reward_meter_std": 0.2714572250843048, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 29 |
+
{"timestamp_utc": "2026-04-11T15:49:18Z", "mode": "train", "global_step": 29, "epoch": 0.0011198640716713006, "loss": -0.0165, "grad_norm": 23.189964294433594, "learning_rate": 9.72e-06, "num_tokens": 54344.0, "completions/mean_length": 43.25, "completions/min_length": 34.0, "completions/max_length": 71.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.25, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 71.0, "rewards/meter/mean": 0.6811927556991577, "rewards/meter/std": 0.32601746916770935, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8811927437782288, "reward_std": 0.3260175287723541, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18413126468658447, "sampling/sampling_logp_difference/max": 1.2698702812194824, "sampling/importance_sampling_ratio/min": 0.2808680534362793, "sampling/importance_sampling_ratio/mean": 1.0205129384994507, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8857728093862534, "entropy": 2.0946053713560104, "clip_ratio/low_mean": 0.07247674837708473, "clip_ratio/low_min": 0.07247674837708473, "clip_ratio/high_mean": 0.15308464504778385, "clip_ratio/high_max": 0.15308464504778385, "clip_ratio/region_mean": 0.22556139342486858, "reward_total_mean": 0.8811927437782288, "reward_meter_mean": 0.6811927556991577, "reward_meter_std": 0.32601746916770935, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 30 |
+
{"timestamp_utc": "2026-04-11T15:49:27Z", "mode": "train", "global_step": 30, "epoch": 0.0011584800741427247, "loss": -0.1499, "grad_norm": 9.213912963867188, "learning_rate": 9.71e-06, "num_tokens": 57126.0, "completions/mean_length": 146.75, "completions/min_length": 71.0, "completions/max_length": 234.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 146.75, "completions/min_terminated_length": 71.0, "completions/max_terminated_length": 234.0, "rewards/meter/mean": 0.3599656820297241, "rewards/meter/std": 0.22725822031497955, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5599656701087952, "reward_std": 0.22725820541381836, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17979256808757782, "sampling/sampling_logp_difference/max": 1.3972516059875488, "sampling/importance_sampling_ratio/min": 0.24727565050125122, "sampling/importance_sampling_ratio/mean": 1.0062932968139648, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0472772866487503, "entropy": 3.4500816762447357, "clip_ratio/low_mean": 0.093433802947402, "clip_ratio/low_min": 0.093433802947402, "clip_ratio/high_mean": 0.1298079490661621, "clip_ratio/high_max": 0.1298079490661621, "clip_ratio/region_mean": 0.2232417520135641, "reward_total_mean": 0.5599656701087952, "reward_meter_mean": 0.3599656820297241, "reward_meter_std": 0.22725822031497955, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 31 |
+
{"timestamp_utc": "2026-04-11T15:49:32Z", "mode": "train", "global_step": 31, "epoch": 0.001197096076614149, "loss": 0.0747, "grad_norm": 20.018091201782227, "learning_rate": 9.7e-06, "num_tokens": 58875.0, "completions/mean_length": 56.625, "completions/min_length": 49.0, "completions/max_length": 64.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 56.625, "completions/min_terminated_length": 49.0, "completions/max_terminated_length": 64.0, "rewards/meter/mean": 0.744500994682312, "rewards/meter/std": 0.19358286261558533, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.9445009827613831, "reward_std": 0.19358289241790771, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.20325925946235657, "sampling/sampling_logp_difference/max": 1.6539866924285889, "sampling/importance_sampling_ratio/min": 0.19128578901290894, "sampling/importance_sampling_ratio/mean": 0.9952805042266846, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5809837579727173, "entropy": 2.848589450120926, "clip_ratio/low_mean": 0.12893840484321117, "clip_ratio/low_min": 0.12893840484321117, "clip_ratio/high_mean": 0.09737317077815533, "clip_ratio/high_max": 0.09737317077815533, "clip_ratio/region_mean": 0.2263115756213665, "reward_total_mean": 0.9445009827613831, "reward_meter_mean": 0.744500994682312, "reward_meter_std": 0.19358286261558533, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 32 |
+
{"timestamp_utc": "2026-04-11T15:49:36Z", "mode": "train", "global_step": 32, "epoch": 0.001235712079085573, "loss": 0.0369, "grad_norm": 27.185209274291992, "learning_rate": 9.69e-06, "num_tokens": 60445.0, "completions/mean_length": 26.25, "completions/min_length": 23.0, "completions/max_length": 34.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 26.25, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 34.0, "rewards/meter/mean": 0.8353732228279114, "rewards/meter/std": 0.3313785493373871, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.0353732109069824, "reward_std": 0.3313785493373871, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1329173594713211, "sampling/sampling_logp_difference/max": 1.3172059059143066, "sampling/importance_sampling_ratio/min": 0.26788273453712463, "sampling/importance_sampling_ratio/mean": 0.9743757247924805, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.1654123663902283, "entropy": 0.9576103650033474, "clip_ratio/low_mean": 0.016304347664117813, "clip_ratio/low_min": 0.016304347664117813, "clip_ratio/high_mean": 0.13171107601374388, "clip_ratio/high_max": 0.13171107601374388, "clip_ratio/region_mean": 0.1480154236778617, "reward_total_mean": 1.0353732109069824, "reward_meter_mean": 0.8353732228279114, "reward_meter_std": 0.3313785493373871, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 33 |
+
{"timestamp_utc": "2026-04-11T15:49:46Z", "mode": "train", "global_step": 33, "epoch": 0.0012743280815569972, "loss": 0.0517, "grad_norm": 8.614265441894531, "learning_rate": 9.68e-06, "num_tokens": 63579.0, "completions/mean_length": 183.75, "completions/min_length": 112.0, "completions/max_length": 272.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 183.75, "completions/min_terminated_length": 112.0, "completions/max_terminated_length": 272.0, "rewards/meter/mean": 0.35417336225509644, "rewards/meter/std": 0.2310473471879959, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.5041733980178833, "reward_std": 0.23062638938426971, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.197601780295372, "sampling/sampling_logp_difference/max": 2.0842056274414062, "sampling/importance_sampling_ratio/min": 0.12440590560436249, "sampling/importance_sampling_ratio/mean": 0.9925354719161987, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3449874222278595, "entropy": 2.8496975153684616, "clip_ratio/low_mean": 0.12502007000148296, "clip_ratio/low_min": 0.12502007000148296, "clip_ratio/high_mean": 0.09342801198363304, "clip_ratio/high_max": 0.09342801198363304, "clip_ratio/region_mean": 0.218448081985116, "reward_total_mean": 0.5041733980178833, "reward_meter_mean": 0.35417336225509644, "reward_meter_std": 0.2310473471879959, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 34 |
+
{"timestamp_utc": "2026-04-11T15:49:51Z", "mode": "train", "global_step": 34, "epoch": 0.0013129440840284213, "loss": 0.1286, "grad_norm": 30.032299041748047, "learning_rate": 9.67e-06, "num_tokens": 64938.0, "completions/mean_length": 24.875, "completions/min_length": 13.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.875, "completions/min_terminated_length": 13.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.3193066716194153, "rewards/meter/std": 0.41723787784576416, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.5193066596984863, "reward_std": 0.4172378480434418, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1874523013830185, "sampling/sampling_logp_difference/max": 2.396888256072998, "sampling/importance_sampling_ratio/min": 0.09100068360567093, "sampling/importance_sampling_ratio/mean": 0.9753797650337219, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.15890334546566, "entropy": 0.7105261906981468, "clip_ratio/low_mean": 0.08128766575828195, "clip_ratio/low_min": 0.08128766575828195, "clip_ratio/high_mean": 0.0798872783780098, "clip_ratio/high_max": 0.0798872783780098, "clip_ratio/region_mean": 0.16117494413629174, "reward_total_mean": 0.5193066596984863, "reward_meter_mean": 0.3193066716194153, "reward_meter_std": 0.41723787784576416, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 35 |
+
{"timestamp_utc": "2026-04-11T15:49:56Z", "mode": "train", "global_step": 35, "epoch": 0.0013515600864998456, "loss": 0.1448, "grad_norm": 15.259716987609863, "learning_rate": 9.66e-06, "num_tokens": 66585.0, "completions/mean_length": 44.875, "completions/min_length": 34.0, "completions/max_length": 65.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 44.875, "completions/min_terminated_length": 34.0, "completions/max_terminated_length": 65.0, "rewards/meter/mean": 0.5695876479148865, "rewards/meter/std": 0.4217164218425751, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.7695876359939575, "reward_std": 0.4217164218425751, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1511761099100113, "sampling/sampling_logp_difference/max": 1.1148014068603516, "sampling/importance_sampling_ratio/min": 0.3279803991317749, "sampling/importance_sampling_ratio/mean": 1.0017306804656982, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.9157908409833908, "entropy": 2.152104765176773, "clip_ratio/low_mean": 0.08943439181894064, "clip_ratio/low_min": 0.08943439181894064, "clip_ratio/high_mean": 0.07368538342416286, "clip_ratio/high_max": 0.07368538342416286, "clip_ratio/region_mean": 0.1631197752431035, "reward_total_mean": 0.7695876359939575, "reward_meter_mean": 0.5695876479148865, "reward_meter_std": 0.4217164218425751, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 36 |
+
{"timestamp_utc": "2026-04-11T15:50:14Z", "mode": "train", "global_step": 36, "epoch": 0.0013901760889712697, "loss": 0.1341, "grad_norm": 4.0301995277404785, "learning_rate": 9.65e-06, "num_tokens": 72074.0, "completions/mean_length": 475.125, "completions/min_length": 352.0, "completions/max_length": 617.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 475.125, "completions/min_terminated_length": 352.0, "completions/max_terminated_length": 617.0, "rewards/meter/mean": 0.6860631704330444, "rewards/meter/std": 0.12680380046367645, "rewards/exact_count_bonus/mean": 0.25, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.7360631227493286, "reward_std": 0.0960330218076706, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13769172132015228, "sampling/sampling_logp_difference/max": 1.4808261394500732, "sampling/importance_sampling_ratio/min": 0.2274497151374817, "sampling/importance_sampling_ratio/mean": 1.0008128881454468, "sampling/importance_sampling_ratio/max": 2.0, "kl": 0.7125953063368797, "entropy": 4.256982505321503, "clip_ratio/low_mean": 0.059955086559057236, "clip_ratio/low_min": 0.059955086559057236, "clip_ratio/high_mean": 0.11802326329052448, "clip_ratio/high_max": 0.11802326329052448, "clip_ratio/region_mean": 0.17797834984958172, "reward_total_mean": 0.7360631227493286, "reward_meter_mean": 0.6860631704330444, "reward_meter_std": 0.12680380046367645, "reward_exact_count_bonus_mean": 0.25, "reward_exact_count_bonus_std": 0.4629100561141968, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 37 |
+
{"timestamp_utc": "2026-04-11T15:50:20Z", "mode": "train", "global_step": 37, "epoch": 0.0014287920914426938, "loss": 0.121, "grad_norm": 15.153329849243164, "learning_rate": 9.640000000000001e-06, "num_tokens": 73927.0, "completions/mean_length": 61.625, "completions/min_length": 46.0, "completions/max_length": 83.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 61.625, "completions/min_terminated_length": 46.0, "completions/max_terminated_length": 83.0, "rewards/meter/mean": 0.6834455728530884, "rewards/meter/std": 0.37012818455696106, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8834455609321594, "reward_std": 0.37012818455696106, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18970002233982086, "sampling/sampling_logp_difference/max": 1.4087400436401367, "sampling/importance_sampling_ratio/min": 0.32161828875541687, "sampling/importance_sampling_ratio/mean": 0.9819008708000183, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.5459181070327759, "entropy": 3.07864186167717, "clip_ratio/low_mean": 0.04557228833436966, "clip_ratio/low_min": 0.04557228833436966, "clip_ratio/high_mean": 0.1836030725389719, "clip_ratio/high_max": 0.1836030725389719, "clip_ratio/region_mean": 0.22917536087334156, "reward_total_mean": 0.8834455609321594, "reward_meter_mean": 0.6834455728530884, "reward_meter_std": 0.37012818455696106, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 38 |
+
{"timestamp_utc": "2026-04-11T15:50:24Z", "mode": "train", "global_step": 38, "epoch": 0.001467408093914118, "loss": 0.017, "grad_norm": 23.363237380981445, "learning_rate": 9.630000000000001e-06, "num_tokens": 75437.0, "completions/mean_length": 25.75, "completions/min_length": 23.0, "completions/max_length": 31.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 25.75, "completions/min_terminated_length": 23.0, "completions/max_terminated_length": 31.0, "rewards/meter/mean": 0.903087854385376, "rewards/meter/std": 0.1558901071548462, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1030879020690918, "reward_std": 0.15589012205600739, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15382516384124756, "sampling/sampling_logp_difference/max": 0.9279258251190186, "sampling/importance_sampling_ratio/min": 0.4016261696815491, "sampling/importance_sampling_ratio/mean": 1.0037627220153809, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7069246247410774, "entropy": 1.842569574713707, "clip_ratio/low_mean": 0.042173911817371845, "clip_ratio/low_min": 0.042173911817371845, "clip_ratio/high_mean": 0.10113142617046833, "clip_ratio/high_max": 0.10113142617046833, "clip_ratio/region_mean": 0.14330533798784018, "reward_total_mean": 1.1030879020690918, "reward_meter_mean": 0.903087854385376, "reward_meter_std": 0.1558901071548462, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 39 |
+
{"timestamp_utc": "2026-04-11T15:50:30Z", "mode": "train", "global_step": 39, "epoch": 0.0015060240963855422, "loss": 0.0859, "grad_norm": 18.175195693969727, "learning_rate": 9.620000000000001e-06, "num_tokens": 77344.0, "completions/mean_length": 74.375, "completions/min_length": 61.0, "completions/max_length": 84.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 74.375, "completions/min_terminated_length": 61.0, "completions/max_terminated_length": 84.0, "rewards/meter/mean": 0.4274764358997345, "rewards/meter/std": 0.42150962352752686, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6274764537811279, "reward_std": 0.42150962352752686, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.16143742203712463, "sampling/sampling_logp_difference/max": 1.6314496994018555, "sampling/importance_sampling_ratio/min": 0.19564573466777802, "sampling/importance_sampling_ratio/mean": 0.9747925996780396, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.829672947525978, "entropy": 1.1887717097997665, "clip_ratio/low_mean": 0.08287477679550648, "clip_ratio/low_min": 0.08287477679550648, "clip_ratio/high_mean": 0.07904236763715744, "clip_ratio/high_max": 0.07904236763715744, "clip_ratio/region_mean": 0.16191714443266392, "reward_total_mean": 0.6274764537811279, "reward_meter_mean": 0.4274764358997345, "reward_meter_std": 0.42150962352752686, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 40 |
+
{"timestamp_utc": "2026-04-11T15:50:36Z", "mode": "train", "global_step": 40, "epoch": 0.0015446400988569664, "loss": 0.0531, "grad_norm": 9.616868019104004, "learning_rate": 9.610000000000001e-06, "num_tokens": 79483.0, "completions/mean_length": 89.375, "completions/min_length": 72.0, "completions/max_length": 123.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 89.375, "completions/min_terminated_length": 72.0, "completions/max_terminated_length": 123.0, "rewards/meter/mean": 0.9042049646377563, "rewards/meter/std": 0.14156009256839752, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1042048931121826, "reward_std": 0.1415601372718811, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1670140027999878, "sampling/sampling_logp_difference/max": 2.528575897216797, "sampling/importance_sampling_ratio/min": 0.07977253943681717, "sampling/importance_sampling_ratio/mean": 0.9803248643875122, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.3452838510274887, "entropy": 3.59776571393013, "clip_ratio/low_mean": 0.051804156973958015, "clip_ratio/low_min": 0.051804156973958015, "clip_ratio/high_mean": 0.1951428446918726, "clip_ratio/high_max": 0.1951428446918726, "clip_ratio/region_mean": 0.2469470016658306, "reward_total_mean": 1.1042048931121826, "reward_meter_mean": 0.9042049646377563, "reward_meter_std": 0.14156009256839752, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 41 |
+
{"timestamp_utc": "2026-04-11T15:50:48Z", "mode": "train", "global_step": 41, "epoch": 0.0015832561013283905, "loss": 0.1829, "grad_norm": 5.928605556488037, "learning_rate": 9.600000000000001e-06, "num_tokens": 83315.0, "completions/mean_length": 264.0, "completions/min_length": 215.0, "completions/max_length": 365.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 264.0, "completions/min_terminated_length": 215.0, "completions/max_terminated_length": 365.0, "rewards/meter/mean": 0.6348835229873657, "rewards/meter/std": 0.3371293842792511, "rewards/exact_count_bonus/mean": 0.75, "rewards/exact_count_bonus/std": 0.4629100561141968, "reward": 0.784883439540863, "reward_std": 0.3861341178417206, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.15597331523895264, "sampling/sampling_logp_difference/max": 1.3521299362182617, "sampling/importance_sampling_ratio/min": 0.3288605809211731, "sampling/importance_sampling_ratio/mean": 0.9977389574050903, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.0418922081589699, "entropy": 3.7767926454544067, "clip_ratio/low_mean": 0.06170275993645191, "clip_ratio/low_min": 0.06170275993645191, "clip_ratio/high_mean": 0.14350228570401669, "clip_ratio/high_max": 0.14350228570401669, "clip_ratio/region_mean": 0.2052050456404686, "reward_total_mean": 0.784883439540863, "reward_meter_mean": 0.6348835229873657, "reward_meter_std": 0.3371293842792511, "reward_exact_count_bonus_mean": 0.75, "reward_exact_count_bonus_std": 0.4629100561141968, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 42 |
+
{"timestamp_utc": "2026-04-11T15:50:54Z", "mode": "train", "global_step": 42, "epoch": 0.0016218721037998146, "loss": -0.0513, "grad_norm": 18.00514030456543, "learning_rate": 9.59e-06, "num_tokens": 85196.0, "completions/mean_length": 65.125, "completions/min_length": 48.0, "completions/max_length": 88.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 65.125, "completions/min_terminated_length": 48.0, "completions/max_terminated_length": 88.0, "rewards/meter/mean": 0.43711739778518677, "rewards/meter/std": 0.3802741467952728, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6121174097061157, "reward_std": 0.3986560106277466, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17133519053459167, "sampling/sampling_logp_difference/max": 1.1742663383483887, "sampling/importance_sampling_ratio/min": 0.3090456426143646, "sampling/importance_sampling_ratio/mean": 0.9917340874671936, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.2616077810525894, "entropy": 3.121804028749466, "clip_ratio/low_mean": 0.11971818003803492, "clip_ratio/low_min": 0.11971818003803492, "clip_ratio/high_mean": 0.08258083090186119, "clip_ratio/high_max": 0.08258083090186119, "clip_ratio/region_mean": 0.2022990109398961, "reward_total_mean": 0.6121174097061157, "reward_meter_mean": 0.43711739778518677, "reward_meter_std": 0.3802741467952728, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 43 |
+
{"timestamp_utc": "2026-04-11T15:50:59Z", "mode": "train", "global_step": 43, "epoch": 0.0016604881062712389, "loss": -0.0307, "grad_norm": 16.364044189453125, "learning_rate": 9.58e-06, "num_tokens": 86814.0, "completions/mean_length": 47.25, "completions/min_length": 30.0, "completions/max_length": 62.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 47.25, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 62.0, "rewards/meter/mean": 0.45132896304130554, "rewards/meter/std": 0.45054808259010315, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6513289213180542, "reward_std": 0.45054808259010315, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19375590980052948, "sampling/sampling_logp_difference/max": 1.3071508407592773, "sampling/importance_sampling_ratio/min": 0.2705899178981781, "sampling/importance_sampling_ratio/mean": 0.996967077255249, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4401512742042542, "entropy": 2.3568301051855087, "clip_ratio/low_mean": 0.09967258013784885, "clip_ratio/low_min": 0.09967258013784885, "clip_ratio/high_mean": 0.1285256128758192, "clip_ratio/high_max": 0.1285256128758192, "clip_ratio/region_mean": 0.22819819301366806, "reward_total_mean": 0.6513289213180542, "reward_meter_mean": 0.45132896304130554, "reward_meter_std": 0.45054808259010315, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 44 |
+
{"timestamp_utc": "2026-04-11T15:51:04Z", "mode": "train", "global_step": 44, "epoch": 0.001699104108742663, "loss": 0.0607, "grad_norm": 23.743331909179688, "learning_rate": 9.57e-06, "num_tokens": 88539.0, "completions/mean_length": 43.625, "completions/min_length": 40.0, "completions/max_length": 48.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 48.0, "rewards/meter/mean": 0.6140779256820679, "rewards/meter/std": 0.3976222574710846, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.8140779733657837, "reward_std": 0.3976222574710846, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.1653001308441162, "sampling/sampling_logp_difference/max": 3.035217761993408, "sampling/importance_sampling_ratio/min": 0.04806419834494591, "sampling/importance_sampling_ratio/mean": 0.9870915412902832, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.507635399699211, "entropy": 0.9922648519277573, "clip_ratio/low_mean": 0.027404863387346268, "clip_ratio/low_min": 0.027404863387346268, "clip_ratio/high_mean": 0.10046137310564518, "clip_ratio/high_max": 0.10046137310564518, "clip_ratio/region_mean": 0.12786623649299145, "reward_total_mean": 0.8140779733657837, "reward_meter_mean": 0.6140779256820679, "reward_meter_std": 0.3976222574710846, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 45 |
+
{"timestamp_utc": "2026-04-11T15:51:08Z", "mode": "train", "global_step": 45, "epoch": 0.001737720111214087, "loss": 0.1034, "grad_norm": 17.9189510345459, "learning_rate": 9.56e-06, "num_tokens": 90315.0, "completions/mean_length": 38.0, "completions/min_length": 32.0, "completions/max_length": 44.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 38.0, "completions/min_terminated_length": 32.0, "completions/max_terminated_length": 44.0, "rewards/meter/mean": 0.433235764503479, "rewards/meter/std": 0.4090244174003601, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6332358121871948, "reward_std": 0.4090244770050049, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.19081313908100128, "sampling/sampling_logp_difference/max": 1.0551395416259766, "sampling/importance_sampling_ratio/min": 0.3481438159942627, "sampling/importance_sampling_ratio/mean": 0.9787829518318176, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.6995926201343536, "entropy": 2.8701943159103394, "clip_ratio/low_mean": 0.09415853396058083, "clip_ratio/low_min": 0.09415853396058083, "clip_ratio/high_mean": 0.09030752815306187, "clip_ratio/high_max": 0.09030752815306187, "clip_ratio/region_mean": 0.1844660621136427, "reward_total_mean": 0.6332358121871948, "reward_meter_mean": 0.433235764503479, "reward_meter_std": 0.4090244174003601, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 46 |
+
{"timestamp_utc": "2026-04-11T15:51:13Z", "mode": "train", "global_step": 46, "epoch": 0.0017763361136855112, "loss": 0.0292, "grad_norm": 35.99632263183594, "learning_rate": 9.55e-06, "num_tokens": 91656.0, "completions/mean_length": 29.625, "completions/min_length": 22.0, "completions/max_length": 41.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 29.625, "completions/min_terminated_length": 22.0, "completions/max_terminated_length": 41.0, "rewards/meter/mean": 0.087132528424263, "rewards/meter/std": 0.08936810493469238, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.28713253140449524, "reward_std": 0.08936809748411179, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2213541567325592, "sampling/sampling_logp_difference/max": 2.1127161979675293, "sampling/importance_sampling_ratio/min": 0.12090910226106644, "sampling/importance_sampling_ratio/mean": 0.9919323921203613, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.7983057498931885, "entropy": 1.8424254208803177, "clip_ratio/low_mean": 0.14126449823379517, "clip_ratio/low_min": 0.14126449823379517, "clip_ratio/high_mean": 0.08075684309005737, "clip_ratio/high_max": 0.08075684309005737, "clip_ratio/region_mean": 0.22202134132385254, "reward_total_mean": 0.28713253140449524, "reward_meter_mean": 0.087132528424263, "reward_meter_std": 0.08936810493469238, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 47 |
+
{"timestamp_utc": "2026-04-11T15:51:17Z", "mode": "train", "global_step": 47, "epoch": 0.0018149521161569355, "loss": 0.0968, "grad_norm": 36.801151275634766, "learning_rate": 9.54e-06, "num_tokens": 93025.0, "completions/mean_length": 24.125, "completions/min_length": 24.0, "completions/max_length": 25.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 24.125, "completions/min_terminated_length": 24.0, "completions/max_terminated_length": 25.0, "rewards/meter/mean": 0.8150457739830017, "rewards/meter/std": 0.03531856834888458, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.0150457620620728, "reward_std": 0.03531854599714279, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.041083987802267075, "sampling/sampling_logp_difference/max": 1.2403922080993652, "sampling/importance_sampling_ratio/min": 0.2892707586288452, "sampling/importance_sampling_ratio/mean": 0.9999908208847046, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.1884380280971527, "entropy": 0.34495257679373026, "clip_ratio/low_mean": 0.02500000037252903, "clip_ratio/low_min": 0.02500000037252903, "clip_ratio/high_mean": 0.0052083334885537624, "clip_ratio/high_max": 0.0052083334885537624, "clip_ratio/region_mean": 0.030208333861082792, "reward_total_mean": 1.0150457620620728, "reward_meter_mean": 0.8150457739830017, "reward_meter_std": 0.03531856834888458, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 48 |
+
{"timestamp_utc": "2026-04-11T15:51:23Z", "mode": "train", "global_step": 48, "epoch": 0.0018535681186283596, "loss": 0.0324, "grad_norm": 17.312498092651367, "learning_rate": 9.53e-06, "num_tokens": 95089.0, "completions/mean_length": 87.0, "completions/min_length": 52.0, "completions/max_length": 118.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 87.0, "completions/min_terminated_length": 52.0, "completions/max_terminated_length": 118.0, "rewards/meter/mean": 0.23315757513046265, "rewards/meter/std": 0.06710986793041229, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.4331575632095337, "reward_std": 0.06710987538099289, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.18739104270935059, "sampling/sampling_logp_difference/max": 1.2994401454925537, "sampling/importance_sampling_ratio/min": 0.2726844251155853, "sampling/importance_sampling_ratio/mean": 0.9660635590553284, "sampling/importance_sampling_ratio/max": 2.0, "kl": 2.2894017100334167, "entropy": 2.6435008347034454, "clip_ratio/low_mean": 0.10152440890669823, "clip_ratio/low_min": 0.10152440890669823, "clip_ratio/high_mean": 0.1315415445715189, "clip_ratio/high_max": 0.1315415445715189, "clip_ratio/region_mean": 0.23306595347821712, "reward_total_mean": 0.4331575632095337, "reward_meter_mean": 0.23315757513046265, "reward_meter_std": 0.06710986793041229, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 49 |
+
{"timestamp_utc": "2026-04-11T15:51:28Z", "mode": "train", "global_step": 49, "epoch": 0.0018921841210997837, "loss": 0.0945, "grad_norm": 15.068387031555176, "learning_rate": 9.52e-06, "num_tokens": 96846.0, "completions/mean_length": 43.625, "completions/min_length": 30.0, "completions/max_length": 59.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 43.625, "completions/min_terminated_length": 30.0, "completions/max_terminated_length": 59.0, "rewards/meter/mean": 0.471576452255249, "rewards/meter/std": 0.4784705936908722, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.6715764999389648, "reward_std": 0.4784706234931946, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.17449823021888733, "sampling/sampling_logp_difference/max": 1.1681387424468994, "sampling/importance_sampling_ratio/min": 0.3757398724555969, "sampling/importance_sampling_ratio/mean": 1.0054690837860107, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.4131104350090027, "entropy": 2.8474780917167664, "clip_ratio/low_mean": 0.11420114897191525, "clip_ratio/low_min": 0.11420114897191525, "clip_ratio/high_mean": 0.11550478264689445, "clip_ratio/high_max": 0.11550478264689445, "clip_ratio/region_mean": 0.2297059316188097, "reward_total_mean": 0.6715764999389648, "reward_meter_mean": 0.471576452255249, "reward_meter_std": 0.4784705936908722, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 50 |
+
{"timestamp_utc": "2026-04-11T15:51:36Z", "mode": "train", "global_step": 50, "epoch": 0.0019308001235712078, "loss": 0.0575, "grad_norm": 8.942173957824707, "learning_rate": 9.51e-06, "num_tokens": 99306.0, "completions/mean_length": 130.5, "completions/min_length": 88.0, "completions/max_length": 182.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 130.5, "completions/min_terminated_length": 88.0, "completions/max_terminated_length": 182.0, "rewards/meter/mean": 0.4820150136947632, "rewards/meter/std": 0.2984078824520111, "rewards/exact_count_bonus/mean": 0.875, "rewards/exact_count_bonus/std": 0.3535533845424652, "reward": 0.6570150256156921, "reward_std": 0.32944732904434204, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.167738139629364, "sampling/sampling_logp_difference/max": 1.5406970977783203, "sampling/importance_sampling_ratio/min": 0.2918128967285156, "sampling/importance_sampling_ratio/mean": 0.9979844093322754, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2031668797135353, "entropy": 3.4669112861156464, "clip_ratio/low_mean": 0.07416268065571785, "clip_ratio/low_min": 0.07416268065571785, "clip_ratio/high_mean": 0.14113500528037548, "clip_ratio/high_max": 0.14113500528037548, "clip_ratio/region_mean": 0.21529768593609333, "reward_total_mean": 0.6570150256156921, "reward_meter_mean": 0.4820150136947632, "reward_meter_std": 0.2984078824520111, "reward_exact_count_bonus_mean": 0.875, "reward_exact_count_bonus_std": 0.3535533845424652, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 51 |
+
{"timestamp_utc": "2026-04-11T15:55:14Z", "mode": "eval", "global_step": 50, "epoch": 0.0019308001235712078, "eval_loss": 0.018872085958719254, "eval_runtime": 217.5742, "eval_samples_per_second": 0.478, "eval_steps_per_second": 0.06, "eval_num_tokens": 99306.0, "eval_completions/mean_length": 153.05769230769232, "eval_completions/min_length": 53.07692307692308, "eval_completions/max_length": 320.3076923076923, "eval_completions/clipped_ratio": 0.0, "eval_completions/mean_terminated_length": 153.05769230769232, "eval_completions/min_terminated_length": 53.07692307692308, "eval_completions/max_terminated_length": 320.3076923076923, "eval_rewards/meter/mean": 0.2994799860394918, "eval_rewards/meter/std": 0.28621239931537557, "eval_rewards/exact_count_bonus/mean": 0.7307692307692307, "eval_rewards/exact_count_bonus/std": 0.3936365097761154, "eval_reward": 0.4456338297862273, "eval_reward_std": 0.1990389978656402, "eval_frac_reward_zero_std": 0.0, "eval_sampling/sampling_logp_difference/mean": 0.03544207819952415, "eval_sampling/sampling_logp_difference/max": 0.29305177239271313, "eval_sampling/importance_sampling_ratio/min": 0.7560820785852579, "eval_sampling/importance_sampling_ratio/mean": 0.9996037987562326, "eval_sampling/importance_sampling_ratio/max": 1.2880034492566035, "eval_kl": 1.2056316297787886, "eval_entropy": 3.3071532799647403, "eval_clip_ratio/low_mean": 0.0, "eval_clip_ratio/low_min": 0.0, "eval_clip_ratio/high_mean": 0.0, "eval_clip_ratio/high_max": 0.0, "eval_clip_ratio/region_mean": 0.0, "eval_reward_total_mean": 0.4456338297862273, "eval_reward_meter_mean": 0.2994799860394918, "eval_reward_meter_std": 0.28621239931537557, "eval_reward_exact_count_bonus_mean": 0.7307692307692307, "eval_reward_exact_count_bonus_std": 0.3936365097761154, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 52 |
+
{"timestamp_utc": "2026-04-11T15:55:20Z", "mode": "train", "global_step": 51, "epoch": 0.001969416126042632, "loss": 0.1217, "grad_norm": 22.3005428314209, "learning_rate": 9.5e-06, "num_tokens": 100856.0, "completions/mean_length": 35.75, "completions/min_length": 27.0, "completions/max_length": 43.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 35.75, "completions/min_terminated_length": 27.0, "completions/max_terminated_length": 43.0, "rewards/meter/mean": 0.2650821805000305, "rewards/meter/std": 0.4003763198852539, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 0.46508222818374634, "reward_std": 0.4003763198852539, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.2505285441875458, "sampling/sampling_logp_difference/max": 1.2456424236297607, "sampling/importance_sampling_ratio/min": 0.2877559959888458, "sampling/importance_sampling_ratio/mean": 0.9820157289505005, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.8204183876514435, "entropy": 2.8815866708755493, "clip_ratio/low_mean": 0.15752424113452435, "clip_ratio/low_min": 0.15752424113452435, "clip_ratio/high_mean": 0.061693549156188965, "clip_ratio/high_max": 0.061693549156188965, "clip_ratio/region_mean": 0.2192177902907133, "reward_total_mean": 0.46508222818374634, "reward_meter_mean": 0.2650821805000305, "reward_meter_std": 0.4003763198852539, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
| 53 |
+
{"timestamp_utc": "2026-04-11T15:55:26Z", "mode": "train", "global_step": 52, "epoch": 0.002008032128514056, "loss": 0.0338, "grad_norm": 12.406262397766113, "learning_rate": 9.49e-06, "num_tokens": 102767.0, "completions/mean_length": 64.875, "completions/min_length": 40.0, "completions/max_length": 78.0, "completions/clipped_ratio": 0.0, "completions/mean_terminated_length": 64.875, "completions/min_terminated_length": 40.0, "completions/max_terminated_length": 78.0, "rewards/meter/mean": 0.9629589319229126, "rewards/meter/std": 0.05814887955784798, "rewards/exact_count_bonus/mean": 1.0, "rewards/exact_count_bonus/std": 0.0, "reward": 1.1629588603973389, "reward_std": 0.058148909360170364, "frac_reward_zero_std": 0.0, "sampling/sampling_logp_difference/mean": 0.13121938705444336, "sampling/sampling_logp_difference/max": 0.9225292205810547, "sampling/importance_sampling_ratio/min": 0.39751237630844116, "sampling/importance_sampling_ratio/mean": 1.004599928855896, "sampling/importance_sampling_ratio/max": 2.0, "kl": 1.2337721958756447, "entropy": 2.761542499065399, "clip_ratio/low_mean": 0.013059701770544052, "clip_ratio/low_min": 0.013059701770544052, "clip_ratio/high_mean": 0.16483690962195396, "clip_ratio/high_max": 0.16483690962195396, "clip_ratio/region_mean": 0.17789661139249802, "reward_total_mean": 1.1629588603973389, "reward_meter_mean": 0.9629589319229126, "reward_meter_std": 0.05814887955784798, "reward_exact_count_bonus_mean": 1.0, "reward_exact_count_bonus_std": 0.0, "run_id": "shaer_grpo_20260411_154531", "run_sequence_index": 0}
|
plots/chain_runs.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
-
"run_id": "
|
| 4 |
-
"run_dir": "/root/workspace/Shaer/grpo/outputs/
|
| 5 |
"run_sequence_index": 0,
|
| 6 |
-
"chain_id": "
|
| 7 |
}
|
| 8 |
]
|
|
|
|
| 1 |
[
|
| 2 |
{
|
| 3 |
+
"run_id": "shaer_grpo_20260411_154531",
|
| 4 |
+
"run_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531",
|
| 5 |
"run_sequence_index": 0,
|
| 6 |
+
"chain_id": "shaer_grpo_20260411_154531"
|
| 7 |
}
|
| 8 |
]
|
plots/reward_panels_eval_chain.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
plots/reward_panels_eval_run.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
plots/reward_panels_train_chain.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
plots/reward_panels_train_run.png
CHANGED
|
Git LFS Details
|
|
Git LFS Details
|
plotter.log
CHANGED
|
@@ -1,52 +1,144 @@
|
|
| 1 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 2 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 3 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 4 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 5 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 6 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 7 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 8 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 9 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 10 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 11 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 12 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 13 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 14 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 15 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 16 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 17 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 18 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 19 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 20 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 21 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 22 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 23 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 24 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 25 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 26 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 27 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 28 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 29 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 30 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 31 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 32 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 33 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 34 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 35 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 36 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 37 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 38 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 39 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 40 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 41 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 42 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 43 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 44 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 45 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 46 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 47 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 48 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 49 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 50 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 51 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
| 52 |
-
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 2 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 3 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 4 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 5 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 6 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 7 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 8 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 9 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 10 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 11 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 12 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 13 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 14 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 15 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 16 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 17 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 18 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 19 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 20 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 21 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 22 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 23 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 24 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 25 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 26 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 27 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 28 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 29 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 30 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 31 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 32 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 33 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 34 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 35 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 36 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 37 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 38 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 39 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 40 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 41 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 42 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 43 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 44 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 45 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 46 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 47 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 48 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 49 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 50 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 51 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 52 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 53 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 54 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 55 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 56 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 57 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 58 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 59 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 60 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 61 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 62 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 63 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 64 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 65 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 66 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 67 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 68 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 69 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 70 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 71 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 72 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 73 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 74 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 75 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 76 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 77 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 78 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 79 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 80 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 81 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 82 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 83 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 84 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 85 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 86 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 87 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 88 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 89 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 90 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 91 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 92 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 93 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 94 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 95 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 96 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 97 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 98 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 99 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 100 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 101 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 102 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 103 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 104 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 105 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 106 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 107 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 108 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 109 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 110 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 111 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 112 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 113 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 114 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 115 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 116 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 117 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 118 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 119 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 120 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 121 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 122 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 123 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 124 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 125 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 126 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 127 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 128 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 129 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 130 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 131 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 132 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 133 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 134 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 135 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 136 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 137 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 138 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 139 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 140 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
| 141 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_run.png
|
| 142 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_run.png
|
| 143 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_train_chain.png
|
| 144 |
+
[plot_live_rewards] updated /root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531/plots/reward_panels_eval_chain.png
|
plotter.pid
CHANGED
|
@@ -1 +1 @@
|
|
| 1 |
-
|
|
|
|
| 1 |
+
13974
|
resume_decision.json
CHANGED
|
@@ -1,9 +1,9 @@
|
|
| 1 |
{
|
| 2 |
-
"timestamp_utc": "2026-04-
|
| 3 |
-
"resume_mode": "
|
| 4 |
"requested_spec": "",
|
| 5 |
"result": "fresh",
|
| 6 |
-
"reason": "
|
| 7 |
"local_resume_path": null,
|
| 8 |
"remote_repo": null,
|
| 9 |
"remote_prefix": null,
|
|
|
|
| 1 |
{
|
| 2 |
+
"timestamp_utc": "2026-04-11T15:46:16Z",
|
| 3 |
+
"resume_mode": "fresh",
|
| 4 |
"requested_spec": "",
|
| 5 |
"result": "fresh",
|
| 6 |
+
"reason": "explicit_fresh",
|
| 7 |
"local_resume_path": null,
|
| 8 |
"remote_repo": null,
|
| 9 |
"remote_prefix": null,
|
reward_exact_count_bonus_debug.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
reward_meter_debug.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
runtime_snapshot.json
CHANGED
|
@@ -1,11 +1,11 @@
|
|
| 1 |
{
|
| 2 |
"config_fingerprint": "f1af89e03892f32e677c238fac409c4e9de7ee84de340974e468cc8fb4c82d8b",
|
| 3 |
"lineage": {
|
| 4 |
-
"created_at_utc": "2026-04-
|
| 5 |
-
"run_id": "
|
| 6 |
-
"run_dir": "/root/workspace/Shaer/grpo/outputs/
|
| 7 |
-
"chain_id": "
|
| 8 |
-
"root_run_id": "
|
| 9 |
"parent_run_id": "",
|
| 10 |
"parent_run_dir": "",
|
| 11 |
"run_sequence_index": 0
|
|
|
|
| 1 |
{
|
| 2 |
"config_fingerprint": "f1af89e03892f32e677c238fac409c4e9de7ee84de340974e468cc8fb4c82d8b",
|
| 3 |
"lineage": {
|
| 4 |
+
"created_at_utc": "2026-04-11T15:46:16Z",
|
| 5 |
+
"run_id": "shaer_grpo_20260411_154531",
|
| 6 |
+
"run_dir": "/root/workspace/Shaer/grpo/outputs/train/shaer_grpo_20260411_154531",
|
| 7 |
+
"chain_id": "shaer_grpo_20260411_154531",
|
| 8 |
+
"root_run_id": "shaer_grpo_20260411_154531",
|
| 9 |
"parent_run_id": "",
|
| 10 |
"parent_run_dir": "",
|
| 11 |
"run_sequence_index": 0
|
train.log
CHANGED
|
@@ -1,17 +1,64 @@
|
|
| 1 |
-
2026-04-11
|
| 2 |
-
2026-04-11
|
| 3 |
-
2026-04-11
|
| 4 |
-
2026-04-11
|
| 5 |
-
2026-04-11
|
| 6 |
-
2026-04-11
|
| 7 |
-
2026-04-11
|
| 8 |
-
2026-04-11
|
| 9 |
-
2026-04-11
|
| 10 |
-
2026-04-11
|
| 11 |
-
2026-04-11
|
| 12 |
-
2026-04-11
|
| 13 |
-
2026-04-11
|
| 14 |
-
2026-04-11
|
| 15 |
-
2026-04-11
|
| 16 |
-
2026-04-11
|
| 17 |
-
2026-04-11
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2026-04-11 15:45:38,719 | INFO | train_grpo_train | mode=train
|
| 2 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | dataset_id=Shaer-AI/ashaar-enhanced-desc-baseform-final-sft-lte20-min500-splits-grpo-meter-count-v1 train_size=25896 eval_size=104
|
| 3 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | source_dataset_id=Shaer-AI/ashaar-with-enhanced-descriptions-baseform-final-sft-lte20-min500-splits test_size=204 hard_diagnostic_size=3126
|
| 4 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | active_rewards=['meter', 'exact_count_bonus']
|
| 5 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | reward_weights={'meter': 1.0, 'exact_count_bonus': 0.2}
|
| 6 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | output_repo=Shaer-AI/Shaer-adapters-grpo
|
| 7 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | train_manifest_path=/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/cap_3000/selected_manifest.csv
|
| 8 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | hard_diagnostic_manifest_path=/root/workspace/Shaer/grpo/outputs/curated_meter_count_locked/hard_diagnostic_cap_256/selected_manifest.csv
|
| 9 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | trainable_params=159907840 total_params=7160467456 trainable_ratio=0.022332
|
| 10 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | resume_decision={"timestamp_utc": "2026-04-11T15:46:16Z", "resume_mode": "fresh", "requested_spec": "", "result": "fresh", "reason": "explicit_fresh", "local_resume_path": null, "remote_repo": null, "remote_prefix": null, "compatibility": "unchecked", "config_fingerprint": "f1af89e03892f32e677c238fac409c4e9de7ee84de340974e468cc8fb4c82d8b"}
|
| 11 |
+
2026-04-11 15:46:37,068 | INFO | train_grpo_train | starting trainer.train()
|
| 12 |
+
2026-04-11 15:46:43,821 | INFO | train_grpo_train | metrics_logged mode=train step=1
|
| 13 |
+
2026-04-11 15:46:49,050 | INFO | train_grpo_train | metrics_logged mode=train step=2
|
| 14 |
+
2026-04-11 15:46:53,894 | INFO | train_grpo_train | metrics_logged mode=train step=3
|
| 15 |
+
2026-04-11 15:46:58,234 | INFO | train_grpo_train | metrics_logged mode=train step=4
|
| 16 |
+
2026-04-11 15:47:07,010 | INFO | train_grpo_train | metrics_logged mode=train step=5
|
| 17 |
+
2026-04-11 15:47:12,886 | INFO | train_grpo_train | metrics_logged mode=train step=6
|
| 18 |
+
2026-04-11 15:47:19,507 | INFO | train_grpo_train | metrics_logged mode=train step=7
|
| 19 |
+
2026-04-11 15:47:24,385 | INFO | train_grpo_train | metrics_logged mode=train step=8
|
| 20 |
+
2026-04-11 15:47:29,491 | INFO | train_grpo_train | metrics_logged mode=train step=9
|
| 21 |
+
2026-04-11 15:47:34,521 | INFO | train_grpo_train | metrics_logged mode=train step=10
|
| 22 |
+
2026-04-11 15:47:39,507 | INFO | train_grpo_train | metrics_logged mode=train step=11
|
| 23 |
+
2026-04-11 15:47:44,756 | INFO | train_grpo_train | metrics_logged mode=train step=12
|
| 24 |
+
2026-04-11 15:47:49,781 | INFO | train_grpo_train | metrics_logged mode=train step=13
|
| 25 |
+
2026-04-11 15:47:56,187 | INFO | train_grpo_train | metrics_logged mode=train step=14
|
| 26 |
+
2026-04-11 15:48:00,559 | INFO | train_grpo_train | metrics_logged mode=train step=15
|
| 27 |
+
2026-04-11 15:48:05,149 | INFO | train_grpo_train | metrics_logged mode=train step=16
|
| 28 |
+
2026-04-11 15:48:09,927 | INFO | train_grpo_train | metrics_logged mode=train step=17
|
| 29 |
+
2026-04-11 15:48:14,933 | INFO | train_grpo_train | metrics_logged mode=train step=18
|
| 30 |
+
2026-04-11 15:48:22,268 | INFO | train_grpo_train | metrics_logged mode=train step=19
|
| 31 |
+
2026-04-11 15:48:27,543 | INFO | train_grpo_train | metrics_logged mode=train step=20
|
| 32 |
+
2026-04-11 15:48:34,770 | INFO | train_grpo_train | metrics_logged mode=train step=21
|
| 33 |
+
2026-04-11 15:48:39,565 | INFO | train_grpo_train | metrics_logged mode=train step=22
|
| 34 |
+
2026-04-11 15:48:44,814 | INFO | train_grpo_train | metrics_logged mode=train step=23
|
| 35 |
+
2026-04-11 15:48:50,002 | INFO | train_grpo_train | metrics_logged mode=train step=24
|
| 36 |
+
2026-04-11 15:48:54,907 | INFO | train_grpo_train | metrics_logged mode=train step=25
|
| 37 |
+
2026-04-11 15:49:00,065 | INFO | train_grpo_train | metrics_logged mode=train step=26
|
| 38 |
+
2026-04-11 15:49:07,152 | INFO | train_grpo_train | metrics_logged mode=train step=27
|
| 39 |
+
2026-04-11 15:49:13,430 | INFO | train_grpo_train | metrics_logged mode=train step=28
|
| 40 |
+
2026-04-11 15:49:18,651 | INFO | train_grpo_train | metrics_logged mode=train step=29
|
| 41 |
+
2026-04-11 15:49:27,412 | INFO | train_grpo_train | metrics_logged mode=train step=30
|
| 42 |
+
2026-04-11 15:49:32,534 | INFO | train_grpo_train | metrics_logged mode=train step=31
|
| 43 |
+
2026-04-11 15:49:36,990 | INFO | train_grpo_train | metrics_logged mode=train step=32
|
| 44 |
+
2026-04-11 15:49:46,682 | INFO | train_grpo_train | metrics_logged mode=train step=33
|
| 45 |
+
2026-04-11 15:49:51,110 | INFO | train_grpo_train | metrics_logged mode=train step=34
|
| 46 |
+
2026-04-11 15:49:56,257 | INFO | train_grpo_train | metrics_logged mode=train step=35
|
| 47 |
+
2026-04-11 15:50:14,884 | INFO | train_grpo_train | metrics_logged mode=train step=36
|
| 48 |
+
2026-04-11 15:50:20,385 | INFO | train_grpo_train | metrics_logged mode=train step=37
|
| 49 |
+
2026-04-11 15:50:24,830 | INFO | train_grpo_train | metrics_logged mode=train step=38
|
| 50 |
+
2026-04-11 15:50:30,368 | INFO | train_grpo_train | metrics_logged mode=train step=39
|
| 51 |
+
2026-04-11 15:50:36,747 | INFO | train_grpo_train | metrics_logged mode=train step=40
|
| 52 |
+
2026-04-11 15:50:48,626 | INFO | train_grpo_train | metrics_logged mode=train step=41
|
| 53 |
+
2026-04-11 15:50:54,230 | INFO | train_grpo_train | metrics_logged mode=train step=42
|
| 54 |
+
2026-04-11 15:50:59,275 | INFO | train_grpo_train | metrics_logged mode=train step=43
|
| 55 |
+
2026-04-11 15:51:04,050 | INFO | train_grpo_train | metrics_logged mode=train step=44
|
| 56 |
+
2026-04-11 15:51:08,734 | INFO | train_grpo_train | metrics_logged mode=train step=45
|
| 57 |
+
2026-04-11 15:51:13,339 | INFO | train_grpo_train | metrics_logged mode=train step=46
|
| 58 |
+
2026-04-11 15:51:17,635 | INFO | train_grpo_train | metrics_logged mode=train step=47
|
| 59 |
+
2026-04-11 15:51:23,855 | INFO | train_grpo_train | metrics_logged mode=train step=48
|
| 60 |
+
2026-04-11 15:51:28,847 | INFO | train_grpo_train | metrics_logged mode=train step=49
|
| 61 |
+
2026-04-11 15:51:36,448 | INFO | train_grpo_train | metrics_logged mode=train step=50
|
| 62 |
+
2026-04-11 15:55:14,027 | INFO | train_grpo_train | metrics_logged mode=eval step=50
|
| 63 |
+
2026-04-11 15:55:20,835 | INFO | train_grpo_train | metrics_logged mode=train step=51
|
| 64 |
+
2026-04-11 15:55:26,558 | INFO | train_grpo_train | metrics_logged mode=train step=52
|
train.pid
CHANGED
|
@@ -1 +1 @@
|
|
| 1 |
-
|
|
|
|
| 1 |
+
13972
|
train_stdout.log
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 7441
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ffc610155bd444b06e4c0c8b5a2e3f293629ece612530c26cf35dfeb7b45c5b3
|
| 3 |
size 7441
|
watcher.pid
CHANGED
|
@@ -1 +1 @@
|
|
| 1 |
-
|
|
|
|
| 1 |
+
13973
|