Pacific-i64 commited on
Commit
f1db031
·
verified ·
1 Parent(s): aa6ffa5

Add files using upload-large-folder tool

Browse files
checkpoints/interrupted_8156/checkpoint.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d02caffb1317cdce768f9eb85e1c36b78cffb2f4af264c33f3b367be1e6a7305
3
+ size 813577067
checkpoints/interrupted_8156/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:448d7e136e95a788259d774733fc6d2658e258503bf542840238f142c8b10ad3
3
+ size 813515200
checkpoints/interrupted_8156/optimizer_rank0.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5ca6d614da79d50e94b111fe737dcfdc88ee12f1e77e53c033aacc979077b564
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank1.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ef4ebbfb4aa8144794c68b7a25c6a123971f9233b3b6d8a4352871bd96dbd158
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank2.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2aae73303bc8ab802fbb19edd851b421ad09841f10787928cd0ba61b06b9de3b
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank3.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0fd8fcaab104d46ed115d6e7c0c2ca6d1a04836b7a25352bda3c56736c88d7e6
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank4.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b10fd7490e9d95e1e8ad348aaa5f8e8462e2c57adf5aaaaa458dd11240cbdb2d
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank5.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a0d806b77e901d948f86b2b3a8aa2bc1a2670ec0306f851acbf724e57e58bae
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank6.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0616dac15f34b56335cd27a182aa052510e98e3eaf552a8a830ff7c7c30f37c6
3
+ size 1609755607
checkpoints/interrupted_8156/optimizer_rank7.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6edf94da5635ae4279eae58ddede712567f9ac6e3585c98c21e27b05a09295bd
3
+ size 1609755607
config.json CHANGED
@@ -1,3 +1,101 @@
1
  {
2
- "view_time": 0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  }
 
1
  {
2
+ "hidden_size": 896,
3
+ "num_hidden_layers": 16,
4
+ "intermediate_size": 256,
5
+ "vocab_size": 32000,
6
+ "num_attention_heads": 14,
7
+ "num_key_value_heads": 2,
8
+ "attention_type": "gqa",
9
+ "attention_dropout": 0.0,
10
+ "use_qk_norm": true,
11
+ "sliding_window": null,
12
+ "causal_conv_kernel_size": 4,
13
+ "causal_conv_dilation_cycle": 8,
14
+ "causal_state_rank": 16,
15
+ "causal_context_gate_init": 1.0,
16
+ "causal_contextual_mix_init": 0.0,
17
+ "causal_context_fusion_size": 0,
18
+ "causal_stable_delta": false,
19
+ "causal_delta_chunk_size": 512,
20
+ "causal_delta_timescales": 1,
21
+ "causal_delta_collision_normalized": false,
22
+ "causal_delta_lexical_values": false,
23
+ "causal_delta_lexical_forge": false,
24
+ "causal_delta_occurrence_address": false,
25
+ "tr_mha_num_experts": 4,
26
+ "tr_mha_adapter_rank": 8,
27
+ "tr_mha_top_k": 2,
28
+ "tr_mha_adapter_gate_init": 0.1,
29
+ "tr_mha_id_primary_logit": 2.0,
30
+ "tr_mha_id_secondary_logit": 1.0,
31
+ "tr_mha_id_other_logit": -2.0,
32
+ "tr_mha_verifier_gate_init": 0.1,
33
+ "tr_mha_verifier_temperature": 1.0,
34
+ "tr_mha_targets": "qv",
35
+ "max_position_embeddings": 2048,
36
+ "rope_theta": 10000.0,
37
+ "rope_type": "standard",
38
+ "rope_fraction": 1.0,
39
+ "mlp_type": "tr_hash_engine",
40
+ "hidden_act": "silu",
41
+ "num_experts": 4,
42
+ "expert_initialization": "gpt_normal",
43
+ "token_frequencies": null,
44
+ "routing_strategy": "token_id_multi_hash",
45
+ "route_hash_count": 2,
46
+ "lsh_routing": false,
47
+ "lsh_bits": 0,
48
+ "lsh_from_layer": 0,
49
+ "lsh_threshold_mode": "zero",
50
+ "shared_expert": true,
51
+ "shared_intermediate_size": 3072,
52
+ "shared_expert_chunk_tokens": 0,
53
+ "use_shared_routed_gates": false,
54
+ "shared_gate_init": 1.0,
55
+ "routed_gate_init": 1.0,
56
+ "shared_output_scale": 1.0,
57
+ "routed_output_scale": 2.0,
58
+ "shared_output_scale_first_layer": null,
59
+ "shared_output_scale_last_layer": null,
60
+ "routed_output_scale_first_layer": null,
61
+ "routed_output_scale_last_layer": null,
62
+ "top_k": 2,
63
+ "top_k_primary_weight": 0.5,
64
+ "learn_hash_pair_gates": false,
65
+ "hash_pair_gate_init": 0.5,
66
+ "learn_hash_channel_modulation": false,
67
+ "hash_channel_scale_init": 0.0,
68
+ "static_expert_capacity": false,
69
+ "use_custom_kernels": "auto",
70
+ "collect_moe_telemetry": false,
71
+ "use_cggr": "auto",
72
+ "active_num_experts": null,
73
+ "active_expert_width": null,
74
+ "lexical_object_rank": 16,
75
+ "lexical_object_gate_init": 0.1,
76
+ "tie_lexical_object_embeddings": false,
77
+ "micro_num_experts": 4,
78
+ "micro_expert_width": 16,
79
+ "micro_expert_gate_init": 0.1,
80
+ "lexical_gqa_rank": 16,
81
+ "lexical_gqa_gate_init": 0.0,
82
+ "lexical_gqa_use_token_code": true,
83
+ "lexical_key_gate_init": 0.05,
84
+ "lexical_zipf_path": null,
85
+ "lexical_zipf_mode": "uniform",
86
+ "lexical_zipf_alpha": 0.25,
87
+ "lexical_zipf_floor": 0.1,
88
+ "lexical_zipf_permutation_seed": 1729,
89
+ "norm_type": "rmsnorm",
90
+ "norm_eps": 1e-06,
91
+ "tie_word_embeddings": true,
92
+ "use_sdpa": true,
93
+ "use_cache": true,
94
+ "is_causal": true,
95
+ "initializer_range": 0.02,
96
+ "use_mup_init": false,
97
+ "use_mup_attn_scale": false,
98
+ "use_mup_output_mult": false,
99
+ "mup_base_width": 256,
100
+ "extra_config": {}
101
  }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:448d7e136e95a788259d774733fc6d2658e258503bf542840238f142c8b10ad3
3
+ size 813515200
model_config.yaml ADDED
@@ -0,0 +1,99 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ active_expert_width: null
2
+ active_num_experts: null
3
+ attention_dropout: 0.0
4
+ attention_type: gqa
5
+ causal_context_fusion_size: 0
6
+ causal_context_gate_init: 1.0
7
+ causal_contextual_mix_init: 0.0
8
+ causal_conv_dilation_cycle: 8
9
+ causal_conv_kernel_size: 4
10
+ causal_delta_chunk_size: 512
11
+ causal_delta_collision_normalized: false
12
+ causal_delta_lexical_forge: false
13
+ causal_delta_lexical_values: false
14
+ causal_delta_occurrence_address: false
15
+ causal_delta_timescales: 1
16
+ causal_stable_delta: false
17
+ causal_state_rank: 16
18
+ collect_moe_telemetry: false
19
+ expert_initialization: gpt_normal
20
+ extra_config: {}
21
+ hash_channel_scale_init: 0.0
22
+ hash_pair_gate_init: 0.5
23
+ hidden_act: silu
24
+ hidden_size: 896
25
+ initializer_range: 0.02
26
+ intermediate_size: 256
27
+ is_causal: true
28
+ learn_hash_channel_modulation: false
29
+ learn_hash_pair_gates: false
30
+ lexical_gqa_gate_init: 0.0
31
+ lexical_gqa_rank: 16
32
+ lexical_gqa_use_token_code: true
33
+ lexical_key_gate_init: 0.05
34
+ lexical_object_gate_init: 0.1
35
+ lexical_object_rank: 16
36
+ lexical_zipf_alpha: 0.25
37
+ lexical_zipf_floor: 0.1
38
+ lexical_zipf_mode: uniform
39
+ lexical_zipf_path: null
40
+ lexical_zipf_permutation_seed: 1729
41
+ lsh_bits: 0
42
+ lsh_from_layer: 0
43
+ lsh_routing: false
44
+ lsh_threshold_mode: zero
45
+ max_position_embeddings: 2048
46
+ micro_expert_gate_init: 0.1
47
+ micro_expert_width: 16
48
+ micro_num_experts: 4
49
+ mlp_type: tr_hash_engine
50
+ mup_base_width: 256
51
+ norm_eps: 1.0e-06
52
+ norm_type: rmsnorm
53
+ num_attention_heads: 14
54
+ num_experts: 4
55
+ num_hidden_layers: 16
56
+ num_key_value_heads: 2
57
+ rope_fraction: 1.0
58
+ rope_theta: 10000.0
59
+ rope_type: standard
60
+ route_hash_count: 2
61
+ routed_gate_init: 1.0
62
+ routed_output_scale: 2.0
63
+ routed_output_scale_first_layer: null
64
+ routed_output_scale_last_layer: null
65
+ routing_strategy: token_id_multi_hash
66
+ shared_expert: true
67
+ shared_expert_chunk_tokens: 0
68
+ shared_gate_init: 1.0
69
+ shared_intermediate_size: 3072
70
+ shared_output_scale: 1.0
71
+ shared_output_scale_first_layer: null
72
+ shared_output_scale_last_layer: null
73
+ sliding_window: null
74
+ static_expert_capacity: false
75
+ tie_lexical_object_embeddings: false
76
+ tie_word_embeddings: true
77
+ token_frequencies: null
78
+ top_k: 2
79
+ top_k_primary_weight: 0.5
80
+ tr_mha_adapter_gate_init: 0.1
81
+ tr_mha_adapter_rank: 8
82
+ tr_mha_id_other_logit: -2.0
83
+ tr_mha_id_primary_logit: 2.0
84
+ tr_mha_id_secondary_logit: 1.0
85
+ tr_mha_num_experts: 4
86
+ tr_mha_targets: qv
87
+ tr_mha_top_k: 2
88
+ tr_mha_verifier_gate_init: 0.1
89
+ tr_mha_verifier_temperature: 1.0
90
+ use_cache: true
91
+ use_cggr: auto
92
+ use_custom_kernels: auto
93
+ use_mup_attn_scale: false
94
+ use_mup_init: false
95
+ use_mup_output_mult: false
96
+ use_qk_norm: true
97
+ use_sdpa: true
98
+ use_shared_routed_gates: false
99
+ vocab_size: 32000
training/metrics.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
training/tensorboard/events.out.tfevents.1787249377.9c6c244bb4b9.2747.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6525ab558b8b684d20c112ca441b9cf9fb027e0c79aabf0b50b0d3f9ecc0359a
3
+ size 1125235
training/tensorboard/events.out.tfevents.1787296464.9c6c244bb4b9.5004.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:65c9bc0a357f486b4ef38ce1caa210ac2c7225e42e1a887f6a08afaaea141302
3
+ size 88
training/training_log.csv ADDED
The diff for this file is too large to render. See raw diff