dimasik87 commited on
Commit
f932571
·
verified ·
1 Parent(s): 81cefb6

Training in progress, step 25, checkpoint

Browse files
last-checkpoint/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fedda0e97e830103c40f700f4873855bbadc6e072a7633e5d19307123b1919fb
3
  size 767856
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cd90ab2ce891868e6fc198c23bcbdce0d8c6ffc7a32620a107291538df25eaf2
3
  size 767856
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:815dbaf33c54ccf6aca94a1a868eea78ef23b0cd589f78171d3beb60101910fd
3
  size 1601338
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f614df1c3150e5e6aa693cec36e7a851bd2c025e86441ff767c2dadadccc8b09
3
  size 1601338
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:067f8b0f85e5ac448631bc4ef4291d838ac65f22fa74b4f23300ecb47979612e
3
  size 14244
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:56e2097e35b44763c2350e2c0f087f7e0938eb18800b471b5a8c2572a33125a6
3
  size 14244
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:9be43866b7a112efbf8125d7bbc11610819a4fb8c7f205bdfffb33dc32734ab8
3
  size 1064
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4521b8db9cc205e54aa606d85e707c024abd2d8ad4a20bec4b2cff365dc59cdf
3
  size 1064
last-checkpoint/trainer_state.json CHANGED
@@ -1,9 +1,9 @@
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
- "epoch": 0.007768247289205373,
5
  "eval_steps": 3,
6
- "global_step": 24,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
@@ -247,6 +247,13 @@
247
  "eval_samples_per_second": 80.166,
248
  "eval_steps_per_second": 40.083,
249
  "step": 24
 
 
 
 
 
 
 
250
  }
251
  ],
252
  "logging_steps": 1,
@@ -261,12 +268,12 @@
261
  "should_evaluate": false,
262
  "should_log": false,
263
  "should_save": true,
264
- "should_training_stop": false
265
  },
266
  "attributes": {}
267
  }
268
  },
269
- "total_flos": 19547654455296.0,
270
  "train_batch_size": 2,
271
  "trial_name": null,
272
  "trial_params": null
 
1
  {
2
  "best_metric": null,
3
  "best_model_checkpoint": null,
4
+ "epoch": 0.00809192425958893,
5
  "eval_steps": 3,
6
+ "global_step": 25,
7
  "is_hyper_param_search": false,
8
  "is_local_process_zero": true,
9
  "is_world_process_zero": true,
 
247
  "eval_samples_per_second": 80.166,
248
  "eval_steps_per_second": 40.083,
249
  "step": 24
250
+ },
251
+ {
252
+ "epoch": 0.00809192425958893,
253
+ "grad_norm": 4.323055267333984,
254
+ "learning_rate": 0.0,
255
+ "loss": 9.4918,
256
+ "step": 25
257
  }
258
  ],
259
  "logging_steps": 1,
 
268
  "should_evaluate": false,
269
  "should_log": false,
270
  "should_save": true,
271
+ "should_training_stop": true
272
  },
273
  "attributes": {}
274
  }
275
  },
276
+ "total_flos": 20362140057600.0,
277
  "train_batch_size": 2,
278
  "trial_name": null,
279
  "trial_params": null