alexkstern commited on
Commit
a7ee513
·
verified ·
1 Parent(s): 0ab70dd

Upload README.md with huggingface_hub

Browse files
Files changed (1) hide show
  1. README.md +139 -0
README.md ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: nanochat
3
+ license: apache-2.0
4
+ tags:
5
+ - nanochat
6
+ - hf_llama
7
+ - vocab_64000
8
+ - pt_owt-gpt2bpe-9B
9
+ - seed_0
10
+ - ppt_c4-gpt2bpe-10B
11
+ - ppt_lr_cosine
12
+ - reinit
13
+ ---
14
+
15
+ # c4_owt_untied_seed0_2026-07-20_18-36-40_388310-owt
16
+
17
+ Trained with [nanochat](https://github.com/karpathy/nanochat). Checkpoint at step **17,196**.
18
+
19
+ **W&B run:** https://wandb.ai/alexksternteam/nca-repro/runs/sewbgrll
20
+
21
+ ## Outcome
22
+
23
+ | metric | value |
24
+ | --- | --- |
25
+ | `step` | 17196 |
26
+ | `smooth_train_loss` | 2.8550471171297045 |
27
+ | `min_objective` | 0.9622754070976698 |
28
+ | `flops_used` | 2.8502607518490427e+19 |
29
+ | `flops_per_token` | 3161456704.0 |
30
+ | `total_training_time` | 58587.57112932205 |
31
+
32
+ ## Training config
33
+
34
+ ```json
35
+ {
36
+ "model_pt": {
37
+ "sequence_len": 1024,
38
+ "vocab_size": 64000,
39
+ "n_layer": 24,
40
+ "n_head": 16,
41
+ "n_embd": 1024,
42
+ "intermediate_size": 4096,
43
+ "rope_theta": 10000.0,
44
+ "attention_dropout": 0.1,
45
+ "weight_tying": false,
46
+ "head_bias": false,
47
+ "initializer_range": 0.02,
48
+ "nanochat_init": false,
49
+ "attn_implementation": "sdpa",
50
+ "explicit_additive_mask": true
51
+ },
52
+ "model_ppt": null,
53
+ "optim": {
54
+ "matrix_lr": 0.0005,
55
+ "embedding_lr": 0.0005,
56
+ "unembedding_lr": 0.0005,
57
+ "weight_decay": 0.0001
58
+ },
59
+ "train": {
60
+ "num_iterations": 1000,
61
+ "eval_every_n_steps": 1000,
62
+ "eval_every_flops_frac": null,
63
+ "save_every_n_steps": null,
64
+ "save_at_steps": null,
65
+ "log_every_n_steps": 10,
66
+ "eval_at_end": true,
67
+ "save_at_end": true,
68
+ "grad_clip": 1.0,
69
+ "ema_beta": 0.9
70
+ },
71
+ "eval": {
72
+ "eval_steps": 16,
73
+ "eval_tokens": 2016000
74
+ },
75
+ "hardware": {
76
+ "device_batch_size": 64,
77
+ "grad_accum_steps": 4,
78
+ "peak_tflops": 2250.0
79
+ },
80
+ "wandb": {
81
+ "enabled": true,
82
+ "notes": "",
83
+ "group": "repro_c4_owt",
84
+ "tags": [
85
+ "hf_llama",
86
+ "vocab_64000",
87
+ "pt_owt-gpt2bpe-9B",
88
+ "seed_0",
89
+ "ppt_c4-gpt2bpe-10B",
90
+ "ppt_lr_cosine",
91
+ "reinit"
92
+ ],
93
+ "entity": null
94
+ },
95
+ "data_pt": "owt-gpt2bpe-9B",
96
+ "data_ppt": "c4-gpt2bpe-10B",
97
+ "extra_eval": [
98
+ "fineweb-gpt2bpe-20B",
99
+ "nca-paper-1024"
100
+ ],
101
+ "train_split": "train",
102
+ "pad_vocab": 64000,
103
+ "lr_kind": "cosine",
104
+ "lr_warmup_ratio": 0.1,
105
+ "lr_warmdown_ratio": 0.0,
106
+ "lr_final_frac": 0.0,
107
+ "ppt_lr": 0.0001,
108
+ "ppt_weight_decay": 0.0,
109
+ "ppt_grad_clip": 0.0,
110
+ "ppt_device_batch_size": 8,
111
+ "ppt_grad_accum_steps": 2,
112
+ "ppt_lr_kind": "cosine",
113
+ "ppt_lr_warmup_ratio": 0.1,
114
+ "ppt_lr_warmdown_ratio": 0.0,
115
+ "ppt_tokens": 164000000,
116
+ "pt_tokens": null,
117
+ "eval_tokens": 2016000,
118
+ "reinit_embed_at_transition": true,
119
+ "reset_optimizer_at_transition": true,
120
+ "depth": null,
121
+ "compile_model": false,
122
+ "model_project": "nca-repro",
123
+ "run_name": "c4_owt_untied_seed0",
124
+ "target_flops": 0.0,
125
+ "alpha_ppt": 0.0,
126
+ "use_measured_flops": true,
127
+ "flops_per_token": null,
128
+ "seed": 0,
129
+ "push_to_hf": true,
130
+ "hf_repo_org": "alexkstern"
131
+ }
132
+ ```
133
+
134
+ ## Files
135
+
136
+ - `model_017196.pt` — model weights (`state_dict`).
137
+ - `meta_017196.json` — training metadata.
138
+ - `config_017196.json` — run config snapshot.
139
+ - `rng_017196.pt` — RNG state (when present).