|
Download README.md from alexkstern/c4_owt_owt50k_2026-06-04_15-00-22_019643-ppt-c4: direct link, hf CLI and curl.
- Browser
- Download file 2.95 kB
-
https://huggingface.co/alexkstern/c4_owt_owt50k_2026-06-04_15-00-22_019643-ppt-c4/resolve/main/README.md
- Command line
-
hf download hf://alexkstern/c4_owt_owt50k_2026-06-04_15-00-22_019643-ppt-c4/README.md
-
curl -L -o README.md https://huggingface.co/alexkstern/c4_owt_owt50k_2026-06-04_15-00-22_019643-ppt-c4/resolve/main/README.md
2.95 kB
| library_name: nanochat | |
| license: apache-2.0 | |
| tags: | |
| - nanochat | |
| - hf_llama | |
| - vocab_64000 | |
| - pt_owt-gpt2bpe-9B | |
| - seed_0 | |
| - ppt_c4-gpt2bpe-10B | |
| - ppt_lr_cosine | |
| - reinit | |
| # c4_owt_owt50k_2026-06-04_15-00-22_019643-ppt-c4 | |
| Trained with [nanochat](https://github.com/karpathy/nanochat). Checkpoint at step **5,009**. | |
| **W&B run:** https://wandb.ai/alexksternteam/nca-repro/runs/xsgly3mc | |
| ## Outcome | |
| | metric | value | | |
| | --- | --- | | |
| | `step` | 5009 | | |
| | `smooth_train_loss` | 4.982999801635742 | | |
| | `min_objective` | 1.4830037126681421 | | |
| | `flops_used` | 5.0509375656165376e+17 | | |
| | `flops_per_token` | 3077308480.0 | | |
| | `total_training_time` | 622.699520111084 | | |
| ## Training config | |
| ```json | |
| { | |
| "model_pt": { | |
| "sequence_len": 1024, | |
| "vocab_size": 50304, | |
| "n_layer": 24, | |
| "n_head": 16, | |
| "n_embd": 1024, | |
| "intermediate_size": 4096, | |
| "rope_theta": 10000.0, | |
| "attention_dropout": 0.1, | |
| "weight_tying": false, | |
| "head_bias": false, | |
| "initializer_range": 0.02, | |
| "attn_implementation": "sdpa" | |
| }, | |
| "model_ppt": null, | |
| "optim": { | |
| "matrix_lr": 0.0001, | |
| "embedding_lr": 0.0001, | |
| "unembedding_lr": 0.0001, | |
| "weight_decay": 0.0 | |
| }, | |
| "train": { | |
| "num_iterations": 1000, | |
| "eval_every_n_steps": 2000, | |
| "save_every_n_steps": null, | |
| "log_every_n_steps": 10, | |
| "eval_at_end": true, | |
| "save_at_end": true, | |
| "grad_clip": 0.0, | |
| "ema_beta": 0.0 | |
| }, | |
| "eval": { | |
| "eval_steps": 62, | |
| "eval_tokens": 2016000 | |
| }, | |
| "hardware": { | |
| "device_batch_size": 8, | |
| "grad_accum_steps": 1, | |
| "peak_tflops": 2250.0 | |
| }, | |
| "wandb": { | |
| "enabled": true, | |
| "notes": "", | |
| "group": "repro_c4_owt", | |
| "tags": [ | |
| "hf_llama", | |
| "vocab_64000", | |
| "pt_owt-gpt2bpe-9B", | |
| "seed_0", | |
| "ppt_c4-gpt2bpe-10B", | |
| "ppt_lr_cosine", | |
| "reinit" | |
| ], | |
| "entity": null | |
| }, | |
| "data_pt": "owt-gpt2bpe-9B", | |
| "data_ppt": "c4-gpt2bpe-10B", | |
| "extra_eval": [ | |
| "fineweb-gpt2bpe-20B", | |
| "nca-paper-1024" | |
| ], | |
| "train_split": "train", | |
| "pad_vocab": 50304, | |
| "lr_kind": "cosine", | |
| "lr_warmup_ratio": 0.1, | |
| "lr_warmdown_ratio": 0.0, | |
| "lr_final_frac": 0.0, | |
| "ppt_lr": 0.0001, | |
| "ppt_weight_decay": 0.0, | |
| "ppt_grad_clip": 0.0, | |
| "ppt_device_batch_size": 8, | |
| "ppt_grad_accum_steps": 1, | |
| "ppt_lr_kind": "cosine", | |
| "ppt_lr_warmup_ratio": 0.1, | |
| "ppt_lr_warmdown_ratio": 0.0, | |
| "ppt_tokens": 164160000, | |
| "pt_tokens": null, | |
| "eval_tokens": 2016000, | |
| "reinit_embed_at_transition": true, | |
| "reset_optimizer_at_transition": true, | |
| "depth": null, | |
| "compile_model": true, | |
| "model_project": "nca-repro", | |
| "run_name": "c4_owt_owt50k", | |
| "target_flops": 0.0, | |
| "alpha_ppt": 0.0, | |
| "use_measured_flops": true, | |
| "seed": 0, | |
| "push_to_hf": true, | |
| "hf_repo_org": "alexkstern" | |
| } | |
| ``` | |
| ## Files | |
| - `model_005009.pt` — model weights (`state_dict`). | |
| - `meta_005009.json` — training metadata. | |
| - `config_005009.json` — run config snapshot. | |
| - `rng_005009.pt` — RNG state (when present). | |