{ "base_model": "Qwen/Qwen3.8-27B", "base_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "best_checkpoint": "/tmp/grug-v2-lora/checkpoint-600", "best_validation_loss": 0.3083864152431488, "dataset_receipt": { "backbone_model": "Qwen/Qwen3.8-27B", "backbone_revision": "1d4bf0f2ff6012fd82039f2fa52739d0dd7c60c0", "groups_disjoint": true, "loss_mask": "last assistant body, reasoning, answer/tool call, EOS; all prompts and history masked", "max_length": 6144, "packing": false, "source_revision": "9bdedcdf86c3b4618f6526879282a5532ee8c690", "splits": { "train": 3466, "validation": 176 }, "statistics": { "overlength_agentic_continue": 86, "overlength_agentic_recovery": 7, "train/agentic_continue/low": 185, "train/agentic_continue/medium": 213, "train/agentic_continue/xhigh": 183, "train/agentic_recovery/low": 65, "train/agentic_recovery/medium": 65, "train/agentic_recovery/xhigh": 65, "train/chat/low": 29, "train/chat/medium": 20, "train/chat/xhigh": 28, "train/coding/low": 269, "train/coding/medium": 302, "train/coding/xhigh": 219, "train/general_qa/low": 24, "train/general_qa/medium": 11, "train/general_qa/xhigh": 23, "train/identity_format/low": 9, "train/identity_format/medium": 10, "train/identity_format/xhigh": 20, "train/math/low": 307, "train/math/medium": 311, "train/math/xhigh": 270, "train/title/low": 92, "train/title/medium": 92, "train/title/xhigh": 92, "train/verified_repository_repair/low": 188, "train/verified_repository_repair/medium": 155, "train/verified_repository_repair/xhigh": 219, "train_supervised_tokens": 968945, "train_tokens": 4476496, "validation/agentic_continue/low": 13, "validation/agentic_continue/medium": 9, "validation/agentic_continue/xhigh": 11, "validation/agentic_recovery/low": 5, "validation/agentic_recovery/medium": 3, "validation/agentic_recovery/xhigh": 10, "validation/chat/low": 2, "validation/chat/xhigh": 1, "validation/coding/low": 17, "validation/coding/medium": 20, "validation/coding/xhigh": 16, "validation/general_qa/low": 1, "validation/general_qa/medium": 1, "validation/identity_format/medium": 1, "validation/math/low": 12, "validation/math/medium": 15, "validation/math/xhigh": 12, "validation/title/low": 4, "validation/title/medium": 4, "validation/title/xhigh": 4, "validation/verified_repository_repair/low": 5, "validation/verified_repository_repair/xhigh": 10, "validation_supervised_tokens": 61905, "validation_tokens": 269482 }, "template_sha256": "cd6193bb4cdc3ee69abd2ae9e15dc05ec3b85a48d161a623bdaae54009c0f995", "tokenizer_source_revision": "3ab073b4bb06dc8a83e819a33485469f32b945ba", "vocabulary_matches_base": true, "vocabulary_sha256": "365c2d8a0c0e72d4bcc5dbc6c2e9330a87dae7ffc77e9fcd9d65739f4d16bfff" }, "epochs_requested": 1.5, "global_steps": 651, "hub_kernels_enabled": true, "initial_validation": { "epoch": 0, "eval_loss": 0.7974646091461182, "eval_model_preparation_time": 0.0405, "eval_runtime": 76.2112, "eval_samples_per_second": 1.26, "eval_steps_per_second": 1.26 }, "learning_rate": 2e-05, "liger_fused_linear_cross_entropy": true, "method": "LoRA r=32 alpha=64", "runtime": { "cuda": "13.0", "gpu": "NVIDIA H200", "peft": "0.20.0", "torch": "2.13.0+cu130", "transformers": "5.16.1" }, "target_modules_regex": "model\\.language_model\\.layers\\.\\d+\\.(?:self_attn\\.(?:q_proj|k_proj|v_proj|o_proj)|linear_attn\\.(?:in_proj_qkv|in_proj_a|in_proj_b|in_proj_z|out_proj)|mlp\\.(?:gate_proj|up_proj|down_proj))", "train_metrics": { "epoch": 1.5005767012687428, "step": 651, "total_flos": 1.1746914808283873e+18, "train_loss": 0.36738806399881563, "train_runtime": 4282.0163, "train_samples_per_second": 1.214, "train_steps_per_second": 0.152 }, "trainable_parameters": 233455616, "training_max_sequence_length": 6144, "training_sequences": 3466 }