# composer_config.yaml — VeRL run config consuming the custom adv_estimator. # # Usage: # PYTHONPATH=/mnt/e/CS/HF/composer-replication-framework/spikes/005-integrated-trainer-skeleton/verl_path \ # python -m verl.trainer.main_ppo --config composer_config.yaml # # (The PYTHONPATH addition makes composer_adv import-and-register at module # load time, so VeRL's adv_estimator dispatch finds "grpo_composer".) # # This is a SKELETON config — paths, sizes, and resource counts are placeholders. # Real v0.2 runs need real paths. algorithm: # Custom estimator from composer_adv.py; registered via @register_adv_est("grpo_composer") adv_estimator: grpo_composer # Channel weights — set either to 0 to ablate that channel alpha_sdpo: 0.1 # SDPO hint-distill (channel 2) beta_replay: 0.05 # N-teacher trace-replay (channel 3) # Standard GRPO knobs kl_ctrl: type: fixed kl_coef: 0.001 use_kl_in_reward: false norm_adv_by_std_in_grpo: true trainer: total_epochs: 1 total_training_steps: 1000 test_freq: 100 save_freq: 200 project_name: composer-replication-v01 experiment_name: qwen3-32b-grpo-composer logger: ['console', 'wandb'] actor_rollout_ref: model: path: /path/to/qwen3-32b # placeholder enable_gradient_checkpointing: true actor: strategy: fsdp2 optim: lr: 1e-6 ppo_mini_batch_size: 64 ppo_micro_batch_size_per_gpu: 4 use_dynamic_bsz: true ulysses_sequence_parallel_size: 1 entropy_coeff: 0.001 rollout: name: vllm n: 8 # group size for GRPO temperature: 1.0 top_p: 0.95 max_response_length: 8192 tensor_model_parallel_size: 4 gpu_memory_utilization: 0.6 max_num_seqs: 64 enforce_eager: false free_cache_engine: false reward_model: enable: false # we use RLVR (reward_func), not RM reward_manager: rule_based # tests-pass / linter / etc. data: train_files: /path/to/train.parquet # placeholder val_files: /path/to/val.parquet prompt_key: prompt max_prompt_length: 2048 max_response_length: 8192 train_batch_size: 64 val_batch_size: 64 # Channel 2 + Channel 3 extras — these are read by the custom rollout worker # (see verl_path/composer_rollout.py once written). They DON'T pass through to # the base GRPO algorithm code — they're consumed by `compute_grpo_composer_advantage`. composer_extras: hint_generator: templates_v01 # registry key in hint_generator.py teachers: - slug: anthropic/claude-opus-4.7 - slug: openai/gpt-5 - slug: deepseek/deepseek-v4-pro trace_replay_voi_gating: enabled: true student_entropy_threshold: 1.5 # bits — only query teachers when student is uncertain reward_hacking_safeguards: sandbox_disable_tools: [find, unzip, strings] sandbox_disable_env_vars: [PYTHONHASHSEED] # for cache-attack mitigation