stefanj0 commited on
Commit
e4363d7
·
verified ·
1 Parent(s): 42f680e

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/adapter_config.json CHANGED
@@ -13,7 +13,7 @@
13
  "layers_pattern": null,
14
  "layers_to_transform": null,
15
  "loftq_config": {},
16
- "lora_alpha": 12,
17
  "lora_bias": false,
18
  "lora_dropout": 0.25,
19
  "megatron_config": null,
@@ -21,17 +21,17 @@
21
  "modules_to_save": null,
22
  "peft_type": "LORA",
23
  "qalora_group_size": 16,
24
- "r": 8,
25
  "rank_pattern": {},
26
  "revision": null,
27
  "target_modules": [
28
- "down_proj",
29
- "k_proj",
30
  "up_proj",
31
- "gate_proj",
 
32
  "v_proj",
33
- "q_proj",
34
- "o_proj"
35
  ],
36
  "target_parameters": null,
37
  "task_type": "SEQ_2_SEQ_LM",
 
13
  "layers_pattern": null,
14
  "layers_to_transform": null,
15
  "loftq_config": {},
16
+ "lora_alpha": 24,
17
  "lora_bias": false,
18
  "lora_dropout": 0.25,
19
  "megatron_config": null,
 
21
  "modules_to_save": null,
22
  "peft_type": "LORA",
23
  "qalora_group_size": 16,
24
+ "r": 12,
25
  "rank_pattern": {},
26
  "revision": null,
27
  "target_modules": [
28
+ "q_proj",
 
29
  "up_proj",
30
+ "k_proj",
31
+ "o_proj",
32
  "v_proj",
33
+ "gate_proj",
34
+ "down_proj"
35
  ],
36
  "target_parameters": null,
37
  "task_type": "SEQ_2_SEQ_LM",
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a3b6cec6bb381d06a610189bd79a8deb709e3c5d2fdf5d819a60d28011decc79
3
- size 5544448
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb7edc022a38beb145ffb51c73bf9826a63972fa7eb7866f4c9659074c1a8c43
3
+ size 8297288
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9da034081b14a2f97a258a7f15ae29da09a4f53994ae505db3cd996ec1942269
3
- size 11256139
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3ff0ff603551669ef970f84399569e9870a0c8597521fcef27f66df3c6f5ef22
3
+ size 16761163
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:97026289a5ff3e49253420f06290674139f61467712a2029e3c1001c12a71fb4
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fc90d0ce81a109e8e9dd4fc59d9ed7b59e19eb969b6e5c7939731348099ada0d
3
  size 14645
last-checkpoint/scaler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:43f4f16c27b661199b56bd3230b24cfb3ac0ec6e2d7124e3005fac49dfb22892
3
  size 1383
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cfc079ff01029f23f0ea697b8a7da7f71ea1510cb9be263645b8bef9dfeb3f70
3
  size 1383
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f00d7d2ea6eff43e89aa523cc15b09467f552729fa90e936a173b939c5db07ba
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:66df8c272ec6ea7fc32240052101ceb1261559eaab8e63c8d4d99dac2464379a
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -1,179 +1,64 @@
1
  {
2
  "best_global_step": 200,
3
- "best_metric": 4.437816143035889,
4
  "best_model_checkpoint": "./t5gemma-math-corrector/checkpoint-200",
5
- "epoch": 3.0,
6
  "eval_steps": 200,
7
- "global_step": 888,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.1693480101608806,
14
- "grad_norm": 1.3522921800613403,
15
- "learning_rate": 5.50561797752809e-05,
16
- "loss": 2.2844,
17
  "step": 50
18
  },
19
  {
20
- "epoch": 0.3386960203217612,
21
- "grad_norm": 0.5189423561096191,
22
- "learning_rate": 9.874843554443054e-05,
23
- "loss": 0.0374,
24
  "step": 100
25
  },
26
  {
27
- "epoch": 0.5080440304826418,
28
- "grad_norm": 0.2796284258365631,
29
- "learning_rate": 9.249061326658323e-05,
30
- "loss": 0.0167,
31
  "step": 150
32
  },
33
  {
34
- "epoch": 0.6773920406435224,
35
- "grad_norm": 0.5644184350967407,
36
- "learning_rate": 8.623279098873592e-05,
37
- "loss": 0.0108,
38
  "step": 200
39
  },
40
  {
41
- "epoch": 0.6773920406435224,
42
- "eval_loss": 4.437816143035889,
43
- "eval_runtime": 10.3595,
44
- "eval_samples_per_second": 50.582,
45
- "eval_steps_per_second": 12.645,
46
  "step": 200
47
- },
48
- {
49
- "epoch": 0.8467400508044031,
50
- "grad_norm": 0.395366907119751,
51
- "learning_rate": 7.997496871088861e-05,
52
- "loss": 0.0073,
53
- "step": 250
54
- },
55
- {
56
- "epoch": 1.0135478408128704,
57
- "grad_norm": 0.14810292422771454,
58
- "learning_rate": 7.371714643304131e-05,
59
- "loss": 0.0079,
60
- "step": 300
61
- },
62
- {
63
- "epoch": 1.182895850973751,
64
- "grad_norm": 0.2394929677248001,
65
- "learning_rate": 6.7459324155194e-05,
66
- "loss": 0.0039,
67
- "step": 350
68
- },
69
- {
70
- "epoch": 1.3522438611346317,
71
- "grad_norm": 0.4444918930530548,
72
- "learning_rate": 6.120150187734669e-05,
73
- "loss": 0.0034,
74
- "step": 400
75
- },
76
- {
77
- "epoch": 1.3522438611346317,
78
- "eval_loss": 7.077005863189697,
79
- "eval_runtime": 10.3846,
80
- "eval_samples_per_second": 50.459,
81
- "eval_steps_per_second": 12.615,
82
- "step": 400
83
- },
84
- {
85
- "epoch": 1.5215918712955123,
86
- "grad_norm": 0.19818231463432312,
87
- "learning_rate": 5.4943679599499376e-05,
88
- "loss": 0.0026,
89
- "step": 450
90
- },
91
- {
92
- "epoch": 1.690939881456393,
93
- "grad_norm": 0.21913927793502808,
94
- "learning_rate": 4.8685857321652064e-05,
95
- "loss": 0.0027,
96
- "step": 500
97
- },
98
- {
99
- "epoch": 1.8602878916172734,
100
- "grad_norm": 0.10620615631341934,
101
- "learning_rate": 4.242803504380476e-05,
102
- "loss": 0.0025,
103
- "step": 550
104
- },
105
- {
106
- "epoch": 2.027095681625741,
107
- "grad_norm": 0.03891622647643089,
108
- "learning_rate": 3.617021276595745e-05,
109
- "loss": 0.002,
110
- "step": 600
111
- },
112
- {
113
- "epoch": 2.027095681625741,
114
- "eval_loss": 7.624501705169678,
115
- "eval_runtime": 10.3997,
116
- "eval_samples_per_second": 50.386,
117
- "eval_steps_per_second": 12.597,
118
- "step": 600
119
- },
120
- {
121
- "epoch": 2.1964436917866217,
122
- "grad_norm": 0.05978698655962944,
123
- "learning_rate": 2.9912390488110137e-05,
124
- "loss": 0.0012,
125
- "step": 650
126
- },
127
- {
128
- "epoch": 2.365791701947502,
129
- "grad_norm": 0.04675230383872986,
130
- "learning_rate": 2.365456821026283e-05,
131
- "loss": 0.0011,
132
- "step": 700
133
- },
134
- {
135
- "epoch": 2.535139712108383,
136
- "grad_norm": 0.056206703186035156,
137
- "learning_rate": 1.739674593241552e-05,
138
- "loss": 0.001,
139
- "step": 750
140
- },
141
- {
142
- "epoch": 2.7044877222692634,
143
- "grad_norm": 0.11963123828172684,
144
- "learning_rate": 1.113892365456821e-05,
145
- "loss": 0.0011,
146
- "step": 800
147
- },
148
- {
149
- "epoch": 2.7044877222692634,
150
- "eval_loss": 7.175710201263428,
151
- "eval_runtime": 10.392,
152
- "eval_samples_per_second": 50.423,
153
- "eval_steps_per_second": 12.606,
154
- "step": 800
155
- },
156
- {
157
- "epoch": 2.873835732430144,
158
- "grad_norm": 0.19036388397216797,
159
- "learning_rate": 4.881101376720902e-06,
160
- "loss": 0.0012,
161
- "step": 850
162
  }
163
  ],
164
  "logging_steps": 50,
165
- "max_steps": 888,
166
  "num_input_tokens_seen": 0,
167
  "num_train_epochs": 3,
168
  "save_steps": 200,
169
  "stateful_callbacks": {
170
  "EarlyStoppingCallback": {
171
  "args": {
172
- "early_stopping_patience": 5,
173
  "early_stopping_threshold": 0.0
174
  },
175
  "attributes": {
176
- "early_stopping_patience_counter": 3
177
  }
178
  },
179
  "TrainerControl": {
@@ -182,12 +67,12 @@
182
  "should_evaluate": false,
183
  "should_log": false,
184
  "should_save": true,
185
- "should_training_stop": true
186
  },
187
  "attributes": {}
188
  }
189
  },
190
- "total_flos": 3379504493887488.0,
191
  "train_batch_size": 4,
192
  "trial_name": null,
193
  "trial_params": null
 
1
  {
2
  "best_global_step": 200,
3
+ "best_metric": 0.6018396615982056,
4
  "best_model_checkpoint": "./t5gemma-math-corrector/checkpoint-200",
5
+ "epoch": 0.6415396952686447,
6
  "eval_steps": 200,
7
+ "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.16038492381716118,
14
+ "grad_norm": 2.080294370651245,
15
+ "learning_rate": 5.212765957446809e-05,
16
+ "loss": 1.8373,
17
  "step": 50
18
  },
19
  {
20
+ "epoch": 0.32076984763432237,
21
+ "grad_norm": 1.3640859127044678,
22
+ "learning_rate": 9.94061757719715e-05,
23
+ "loss": 0.028,
24
  "step": 100
25
  },
26
  {
27
+ "epoch": 0.48115477145148355,
28
+ "grad_norm": 0.47297751903533936,
29
+ "learning_rate": 9.346793349168646e-05,
30
+ "loss": 0.0138,
31
  "step": 150
32
  },
33
  {
34
+ "epoch": 0.6415396952686447,
35
+ "grad_norm": 0.6050183176994324,
36
+ "learning_rate": 8.752969121140144e-05,
37
+ "loss": 0.0085,
38
  "step": 200
39
  },
40
  {
41
+ "epoch": 0.6415396952686447,
42
+ "eval_loss": 0.6018396615982056,
43
+ "eval_runtime": 5.1929,
44
+ "eval_samples_per_second": 50.454,
45
+ "eval_steps_per_second": 12.71,
46
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
47
  }
48
  ],
49
  "logging_steps": 50,
50
+ "max_steps": 936,
51
  "num_input_tokens_seen": 0,
52
  "num_train_epochs": 3,
53
  "save_steps": 200,
54
  "stateful_callbacks": {
55
  "EarlyStoppingCallback": {
56
  "args": {
57
+ "early_stopping_patience": 3,
58
  "early_stopping_threshold": 0.0
59
  },
60
  "attributes": {
61
+ "early_stopping_patience_counter": 0
62
  }
63
  },
64
  "TrainerControl": {
 
67
  "should_evaluate": false,
68
  "should_log": false,
69
  "should_save": true,
70
+ "should_training_stop": false
71
  },
72
  "attributes": {}
73
  }
74
  },
75
+ "total_flos": 773230008729600.0,
76
  "train_batch_size": 4,
77
  "trial_name": null,
78
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8c67ac096c206b245d26f775afa3fb8ffa810038da5c0f227d1b18811d5d602b
3
  size 5905
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a5e19188bdca9ed7ec3f5485c2058b4803117cec11f54ee11e729dcda3bd5b01
3
  size 5905