(0.6B)/(40M-20M)/(32B-predict-revise-k=8,L=512,r=0.03125,k=8)/(lr=0.00015,epochs=2,replay=0.25)/(seed=0)/(val-r=1.0)

Prestar continued-pre-training checkpoint.

  • Checkpoint kind: latest
  • Training step: 510
  • Run name: (0.6B)/(40M-20M)/(32B-predict-revise-k=8,L=512,r=0.03125,k=8)/(lr=0.00015,epochs=2,replay=0.25)/(seed=0)/(val-r=1.0)

Resolved training config

trainee:
  short_name: 0.6B
  model_id: Qwen/Qwen3-0.6B
  dtype: bfloat16
train:
  data:
    documents:
      short_name: 40M-20M
      dataset_id: JackHsieh/statML-arxiv-40M-20M
      split: train
      n_samples: null
      exact_document_length: 4096
    chunking:
      rule: r=0.03125,k=8
      chunk_size: 8
      path: outputs/prestar/chunking/r=0.03125,k=8/40M-20M/train.parquet
    thoughts:
      short_name: 32B-predict-revise-k=8,L=512
      dataset_id: JackHsieh/32B-predict-revise.rule-r-0.03125-k-8.L-512.statml-arxiv
      split: train
      n_samples: null
    replay:
      short_name: full-replay
      dataset_id: JackHsieh/dclm-replay.seq-4096.tokens-32B
      split: train
      n_samples: null
      exact_document_length: 4096
    serving:
      use_thoughts: true
      replay_proportion: 0.25
      thoughts_per_step: 1
      fixed_main_order: true
      fixed_replay_order: true
  enable: true
  docs_per_batch: 32
  total_epochs: 2
  total_steps: null
  grad_clip_norm: 1.0
  seed: 0
  early_stopping:
    enable: false
    mode: null
    metric: null
    patience_evals: null
    min_delta: null
  loss: {}
  execution:
    thoughtless_pass_doc_microbatch: 4
    thoughtful_pass_thought_microbatch: 4
    save_incremental_per_epoch: false
    save_per_token_log_probs: false
    save_chunk_token_snippets: false
  optimizer:
    style: adamw
    lr: 0.00015
    weight_decay: 0.01
    beta1: 0.9
    beta2: 0.95
    eps: 1.0e-08
  lr_scheduler:
    style: cosine
    total_steps: null
    min_lr_ratio: 0.0
    warmup_steps: null
    warmup_ratio: 0.05
    stable_steps: null
    decay_steps: null
eval:
  data:
    documents:
      short_name: 40M-20M
      dataset_id: JackHsieh/statML-arxiv-40M-20M
      split: test
      n_samples: null
      exact_document_length: 4096
    chunking:
      rule: r=1.0,k=8
      chunk_size: 8
      path: outputs/prestar/chunking/r=1.0,k=8/40M-20M/test.parquet
    thoughts:
      short_name: 32B-predict-revise-k=8,L=512
      dataset_id: JackHsieh/32B-predict-revise.rule-r-1.0-k-8.L-512.statml-arxiv
      split: test
      n_samples: null
    serving:
      use_thoughts: true
      thoughts_per_pass: 1
  enable: true
  eval_every_n_steps: null
  eval_every_n_epochs: 2
  eval_on_first_step: false
  execution:
    docs_per_batch: null
    thoughtless_pass_doc_microbatch: 32
    thoughtful_pass_thought_microbatch: 8
    incremental_save_every_n_batches: 0
    save_per_token_log_probs: true
    save_chunk_token_snippets: false
checkpointing:
  resume:
    from_latest: true
    from_path: null
    same_wandb_run: true
  best:
    enable: false
    metric: val-dynamics/val_loss
    direction: min
    max_keep: 1
    includes_resume_state: false
    hub:
      enable: false
      repo_id: null
      private: false
      model_card_template_path: null
  latest:
    enable: true
    max_keep: 1
    includes_resume_state: true
    cadence:
      unit: epochs
      every: 0.25
    hub:
      enable: true
      repo_id: null
      private: false
      model_card_template_path: null
logging:
  flush_every_n_steps: 1
  train_metrics_reduction: aggregate
  file:
    per_batch_metrics_jsonl: true
  wandb:
    enable: true
    project: prestar
    entity: latent-thoughts
    group: thoughtful-CPT-32B-predict-fp32
    name: null
    tags: []
    n_examples_logged: 10
    notes_template_path: configs/prestar/wandb_templates/default.j2
  debug:
    enabled: false
    cross_rank_checks: false
hardware:
  num_gpus: 8
  cpus_per_gpu: 8
  gpu_type: null
  peak_bf16_tflops: null
system:
  vllm: null
  trainee:
    attention_implementation: flash_attention_2
    enable_gradient_checkpointing: false
    enable_activation_offload: false
  fsdp:
    param_offload: false
    optimizer_offload: false
    fsdp_size: -1
    strategy: no_shard
    reshard_after_forward: false
    master_weights_fp32: true
run_name: (0.6B)/(40M-20M)/(32B-predict-revise-k=8,L=512,r=0.03125,k=8)/(lr=0.00015,epochs=2,replay=0.25)/(seed=0)/(val-r=1.0)
outputs_root: outputs/prestar
Downloads last month

-

Downloads are not tracked for this model. How to track
Inference Providers NEW
This model isn't deployed by any Inference Provider. 🙋 Ask for provider support