| kind: nla_model | |
| schema_version: 2 | |
| role: critic | |
| stage: sl | |
| base_checkpoint: /workspace/models/rl-ar | |
| d_model: 3584 | |
| extraction: | |
| injection_scale: null | |
| mse_scale: 59.86651818838306 | |
| tokens: | |
| injection_char: "\u320E" | |
| injection_token_id: 149705 | |
| injection_left_neighbor_id: 29 | |
| injection_right_neighbor_id: 522 | |
| critic_suffix_ids: | |
| - 1318 | |
| - 29 | |
| - 366 | |
| - 1708 | |
| - 29 | |
| prompt_templates: | |
| actor: 'You are a meticulous AI researcher conducting an important investigation | |
| into activation vectors from a language model. Your overall task is to describe | |
| the semantic content of that activation vector. | |
| We will pass the vector enclosed in <concept> tags into your context. You must | |
| then describe that vector as a list of bullet points, one per line, ordered from | |
| the most to the least salient aspect. | |
| Here is the vector: | |
| <concept>{injection_char}</concept> | |
| Please provide the bullet points.' | |
| critic: 'Summary of the following text: <text>{explanation}</text> <summary>' | |
| trained_on: | |
| - /workspace/out/ar_sft_v3.parquet | |
| parent_checkpoints: | |
| - /workspace/models/rl-ar | |
| created_at: '2026-07-01T23:59:14.786251+00:00' | |
| created_by: nla.train_actor.NLAFSDPActor | |
| critic: | |
| extraction_layer_index: 20 | |
| training: | |
| rollout_id: 856 | |
| lr: 2.0e-05 | |
| loss_type: custom_loss | |
| global_batch_size: 256 | |