xummer commited on
Commit
04824d7
·
verified ·
1 Parent(s): 6b2e4bf

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -33,3 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ checkpoint-500/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ checkpoint-846/tokenizer.json filter=lfs diff=lfs merge=lfs -text
38
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,70 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ library_name: peft
3
+ license: other
4
+ base_model: google/gemma-2-9b-it
5
+ tags:
6
+ - base_model:adapter:google/gemma-2-9b-it
7
+ - llama-factory
8
+ - lora
9
+ - transformers
10
+ metrics:
11
+ - accuracy
12
+ pipeline_tag: text-generation
13
+ model-index:
14
+ - name: nli_P2_multi_n500_seed42
15
+ results: []
16
+ ---
17
+
18
+ <!-- This model card has been generated automatically according to the information the Trainer had access to. You
19
+ should probably proofread and complete it, then remove this comment. -->
20
+
21
+ # nli_P2_multi_n500_seed42
22
+
23
+ This model is a fine-tuned version of [google/gemma-2-9b-it](https://huggingface.co/google/gemma-2-9b-it) on the nli_multi_n500_train dataset.
24
+ It achieves the following results on the evaluation set:
25
+ - Loss: 0.2139
26
+ - Accuracy: 0.9458
27
+ - Mcq Accuracy: 0.7311
28
+
29
+ ## Model description
30
+
31
+ More information needed
32
+
33
+ ## Intended uses & limitations
34
+
35
+ More information needed
36
+
37
+ ## Training and evaluation data
38
+
39
+ More information needed
40
+
41
+ ## Training procedure
42
+
43
+ ### Training hyperparameters
44
+
45
+ The following hyperparameters were used during training:
46
+ - learning_rate: 0.0002
47
+ - train_batch_size: 4
48
+ - eval_batch_size: 4
49
+ - seed: 42
50
+ - gradient_accumulation_steps: 4
51
+ - total_train_batch_size: 16
52
+ - optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
53
+ - lr_scheduler_type: cosine
54
+ - lr_scheduler_warmup_steps: 0.1
55
+ - num_epochs: 3.0
56
+
57
+ ### Training results
58
+
59
+ | Training Loss | Epoch | Step | Validation Loss | Accuracy | Mcq Accuracy |
60
+ |:-------------:|:------:|:----:|:---------------:|:--------:|:------------:|
61
+ | 0.0458 | 1.7751 | 500 | 0.1556 | 0.9427 | 0.7222 |
62
+
63
+
64
+ ### Framework versions
65
+
66
+ - PEFT 0.18.1
67
+ - Transformers 5.2.0
68
+ - Pytorch 2.10.0+cu128
69
+ - Datasets 4.0.0
70
+ - Tokenizers 0.22.2
adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "google/gemma-2-9b-it",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "up_proj",
37
+ "v_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25d2961ba873b47b3a0145266448f754de82479dc07e1e2a42d8e25899ec8820
3
+ size 216151256
all_results.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "epoch": 3.0,
3
+ "eval_accuracy": 0.9458148148148149,
4
+ "eval_loss": 0.21385908126831055,
5
+ "eval_mcq_accuracy": 0.7311111111111112,
6
+ "eval_runtime": 26.4009,
7
+ "eval_samples_per_second": 17.045,
8
+ "eval_steps_per_second": 4.28,
9
+ "total_flos": 1.0958550940031386e+17,
10
+ "train_loss": 0.06952556652285546,
11
+ "train_runtime": 2162.4382,
12
+ "train_samples_per_second": 6.243,
13
+ "train_steps_per_second": 0.391
14
+ }
chat_template.jinja ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {{ bos_token }}{% if messages[0]['role'] == 'system' %}{{ raise_exception('System role not supported') }}{% endif %}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% if (message['role'] == 'assistant') %}{% set role = 'model' %}{% else %}{% set role = message['role'] %}{% endif %}{{ '<start_of_turn>' + role + '
2
+ ' + message['content'] | trim + '<end_of_turn>
3
+ ' }}{% endfor %}{% if add_generation_prompt %}{{'<start_of_turn>model
4
+ '}}{% endif %}
checkpoint-500/README.md ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: google/gemma-2-9b-it
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:google/gemma-2-9b-it
7
+ - llama-factory
8
+ - lora
9
+ - transformers
10
+ ---
11
+
12
+ # Model Card for Model ID
13
+
14
+ <!-- Provide a quick summary of what the model is/does. -->
15
+
16
+
17
+
18
+ ## Model Details
19
+
20
+ ### Model Description
21
+
22
+ <!-- Provide a longer summary of what this model is. -->
23
+
24
+
25
+
26
+ - **Developed by:** [More Information Needed]
27
+ - **Funded by [optional]:** [More Information Needed]
28
+ - **Shared by [optional]:** [More Information Needed]
29
+ - **Model type:** [More Information Needed]
30
+ - **Language(s) (NLP):** [More Information Needed]
31
+ - **License:** [More Information Needed]
32
+ - **Finetuned from model [optional]:** [More Information Needed]
33
+
34
+ ### Model Sources [optional]
35
+
36
+ <!-- Provide the basic links for the model. -->
37
+
38
+ - **Repository:** [More Information Needed]
39
+ - **Paper [optional]:** [More Information Needed]
40
+ - **Demo [optional]:** [More Information Needed]
41
+
42
+ ## Uses
43
+
44
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
45
+
46
+ ### Direct Use
47
+
48
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
49
+
50
+ [More Information Needed]
51
+
52
+ ### Downstream Use [optional]
53
+
54
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
55
+
56
+ [More Information Needed]
57
+
58
+ ### Out-of-Scope Use
59
+
60
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
61
+
62
+ [More Information Needed]
63
+
64
+ ## Bias, Risks, and Limitations
65
+
66
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
67
+
68
+ [More Information Needed]
69
+
70
+ ### Recommendations
71
+
72
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
73
+
74
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
75
+
76
+ ## How to Get Started with the Model
77
+
78
+ Use the code below to get started with the model.
79
+
80
+ [More Information Needed]
81
+
82
+ ## Training Details
83
+
84
+ ### Training Data
85
+
86
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
87
+
88
+ [More Information Needed]
89
+
90
+ ### Training Procedure
91
+
92
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
93
+
94
+ #### Preprocessing [optional]
95
+
96
+ [More Information Needed]
97
+
98
+
99
+ #### Training Hyperparameters
100
+
101
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
102
+
103
+ #### Speeds, Sizes, Times [optional]
104
+
105
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
106
+
107
+ [More Information Needed]
108
+
109
+ ## Evaluation
110
+
111
+ <!-- This section describes the evaluation protocols and provides the results. -->
112
+
113
+ ### Testing Data, Factors & Metrics
114
+
115
+ #### Testing Data
116
+
117
+ <!-- This should link to a Dataset Card if possible. -->
118
+
119
+ [More Information Needed]
120
+
121
+ #### Factors
122
+
123
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
124
+
125
+ [More Information Needed]
126
+
127
+ #### Metrics
128
+
129
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
130
+
131
+ [More Information Needed]
132
+
133
+ ### Results
134
+
135
+ [More Information Needed]
136
+
137
+ #### Summary
138
+
139
+
140
+
141
+ ## Model Examination [optional]
142
+
143
+ <!-- Relevant interpretability work for the model goes here -->
144
+
145
+ [More Information Needed]
146
+
147
+ ## Environmental Impact
148
+
149
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
150
+
151
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
152
+
153
+ - **Hardware Type:** [More Information Needed]
154
+ - **Hours used:** [More Information Needed]
155
+ - **Cloud Provider:** [More Information Needed]
156
+ - **Compute Region:** [More Information Needed]
157
+ - **Carbon Emitted:** [More Information Needed]
158
+
159
+ ## Technical Specifications [optional]
160
+
161
+ ### Model Architecture and Objective
162
+
163
+ [More Information Needed]
164
+
165
+ ### Compute Infrastructure
166
+
167
+ [More Information Needed]
168
+
169
+ #### Hardware
170
+
171
+ [More Information Needed]
172
+
173
+ #### Software
174
+
175
+ [More Information Needed]
176
+
177
+ ## Citation [optional]
178
+
179
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
180
+
181
+ **BibTeX:**
182
+
183
+ [More Information Needed]
184
+
185
+ **APA:**
186
+
187
+ [More Information Needed]
188
+
189
+ ## Glossary [optional]
190
+
191
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
192
+
193
+ [More Information Needed]
194
+
195
+ ## More Information [optional]
196
+
197
+ [More Information Needed]
198
+
199
+ ## Model Card Authors [optional]
200
+
201
+ [More Information Needed]
202
+
203
+ ## Model Card Contact
204
+
205
+ [More Information Needed]
206
+ ### Framework versions
207
+
208
+ - PEFT 0.18.1
checkpoint-500/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "google/gemma-2-9b-it",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "up_proj",
37
+ "v_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
checkpoint-500/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a8015f958b88d941dfb41964df7e79a500ad58df6a4c8eb2df646d70ca1ac1ac
3
+ size 216151256
checkpoint-500/chat_template.jinja ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {{ bos_token }}{% if messages[0]['role'] == 'system' %}{{ raise_exception('System role not supported') }}{% endif %}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% if (message['role'] == 'assistant') %}{% set role = 'model' %}{% else %}{% set role = message['role'] %}{% endif %}{{ '<start_of_turn>' + role + '
2
+ ' + message['content'] | trim + '<end_of_turn>
3
+ ' }}{% endfor %}{% if add_generation_prompt %}{{'<start_of_turn>model
4
+ '}}{% endif %}
checkpoint-500/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:394ace002a144ac6ad5486387502f2d36f70c087310c3d907857240c76fcb36e
3
+ size 34362748
checkpoint-500/tokenizer_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "bos_token": "<bos>",
4
+ "clean_up_tokenization_spaces": false,
5
+ "eos_token": "<eos>",
6
+ "extra_special_tokens": [
7
+ "<eos>",
8
+ "<end_of_turn>"
9
+ ],
10
+ "is_local": false,
11
+ "mask_token": "<mask>",
12
+ "model_max_length": 1000000000000000019884624838656,
13
+ "pad_token": "<pad>",
14
+ "padding_side": "right",
15
+ "sp_model_kwargs": {},
16
+ "spaces_between_special_tokens": false,
17
+ "split_special_tokens": false,
18
+ "tokenizer_class": "GemmaTokenizer",
19
+ "unk_token": "<unk>",
20
+ "use_default_system_prompt": false
21
+ }
checkpoint-500/trainer_state.json ADDED
@@ -0,0 +1,394 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 1.775111111111111,
6
+ "eval_steps": 500,
7
+ "global_step": 500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.035555555555555556,
14
+ "grad_norm": 0.7644901871681213,
15
+ "learning_rate": 2.1176470588235296e-05,
16
+ "loss": 0.30520846843719485,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.07111111111111111,
21
+ "grad_norm": 0.43560856580734253,
22
+ "learning_rate": 4.470588235294118e-05,
23
+ "loss": 0.26438264846801757,
24
+ "step": 20
25
+ },
26
+ {
27
+ "epoch": 0.10666666666666667,
28
+ "grad_norm": 0.2563800811767578,
29
+ "learning_rate": 6.823529411764707e-05,
30
+ "loss": 0.19181081056594848,
31
+ "step": 30
32
+ },
33
+ {
34
+ "epoch": 0.14222222222222222,
35
+ "grad_norm": 0.10259224474430084,
36
+ "learning_rate": 9.176470588235295e-05,
37
+ "loss": 0.15193926095962523,
38
+ "step": 40
39
+ },
40
+ {
41
+ "epoch": 0.17777777777777778,
42
+ "grad_norm": 0.1993834525346756,
43
+ "learning_rate": 0.00011529411764705881,
44
+ "loss": 0.13089948892593384,
45
+ "step": 50
46
+ },
47
+ {
48
+ "epoch": 0.21333333333333335,
49
+ "grad_norm": 0.19039888679981232,
50
+ "learning_rate": 0.00013882352941176472,
51
+ "loss": 0.1438336730003357,
52
+ "step": 60
53
+ },
54
+ {
55
+ "epoch": 0.24888888888888888,
56
+ "grad_norm": 0.153082013130188,
57
+ "learning_rate": 0.0001623529411764706,
58
+ "loss": 0.13305349349975587,
59
+ "step": 70
60
+ },
61
+ {
62
+ "epoch": 0.28444444444444444,
63
+ "grad_norm": 0.08913639187812805,
64
+ "learning_rate": 0.00018588235294117648,
65
+ "loss": 0.16499919891357423,
66
+ "step": 80
67
+ },
68
+ {
69
+ "epoch": 0.32,
70
+ "grad_norm": 0.18134109675884247,
71
+ "learning_rate": 0.00019998636639992777,
72
+ "loss": 0.13644593954086304,
73
+ "step": 90
74
+ },
75
+ {
76
+ "epoch": 0.35555555555555557,
77
+ "grad_norm": 0.09949612617492676,
78
+ "learning_rate": 0.00019983303108908946,
79
+ "loss": 0.13039498329162597,
80
+ "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.39111111111111113,
84
+ "grad_norm": 0.12579314410686493,
85
+ "learning_rate": 0.00019950958062149127,
86
+ "loss": 0.1297929048538208,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.4266666666666667,
91
+ "grad_norm": 0.1197328120470047,
92
+ "learning_rate": 0.00019901656615566656,
93
+ "loss": 0.1509210705757141,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.4622222222222222,
98
+ "grad_norm": 0.15709449350833893,
99
+ "learning_rate": 0.00019835482778664425,
100
+ "loss": 0.14251822233200073,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.49777777777777776,
105
+ "grad_norm": 0.2203192412853241,
106
+ "learning_rate": 0.0001975254931144296,
107
+ "loss": 0.1412465214729309,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 0.5333333333333333,
112
+ "grad_norm": 0.32334810495376587,
113
+ "learning_rate": 0.0001965299753225775,
114
+ "loss": 0.1519417643547058,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 0.5688888888888889,
119
+ "grad_norm": 0.1700880080461502,
120
+ "learning_rate": 0.00019536997077013236,
121
+ "loss": 0.13457289934158326,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 0.6044444444444445,
126
+ "grad_norm": 0.17782148718833923,
127
+ "learning_rate": 0.00019404745610103786,
128
+ "loss": 0.14420714378356933,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 0.64,
133
+ "grad_norm": 0.19534672796726227,
134
+ "learning_rate": 0.00019256468487594214,
135
+ "loss": 0.1206929087638855,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 0.6755555555555556,
140
+ "grad_norm": 0.12961238622665405,
141
+ "learning_rate": 0.00019092418373213796,
142
+ "loss": 0.13416168689727784,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 0.7111111111111111,
147
+ "grad_norm": 0.19645075500011444,
148
+ "learning_rate": 0.000189128748078181,
149
+ "loss": 0.13229730129241943,
150
+ "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.7466666666666667,
154
+ "grad_norm": 0.17324262857437134,
155
+ "learning_rate": 0.00018718143733052278,
156
+ "loss": 0.12280937433242797,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.7822222222222223,
161
+ "grad_norm": 0.10764078050851822,
162
+ "learning_rate": 0.0001850855697002753,
163
+ "loss": 0.12583138942718505,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.8177777777777778,
168
+ "grad_norm": 0.13804763555526733,
169
+ "learning_rate": 0.00018284471653898994,
170
+ "loss": 0.13474913835525512,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.8533333333333334,
175
+ "grad_norm": 0.11816427856683731,
176
+ "learning_rate": 0.00018046269625308648,
177
+ "loss": 0.13484352827072144,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 0.8888888888888888,
182
+ "grad_norm": 0.09677577018737793,
183
+ "learning_rate": 0.00017794356779730084,
184
+ "loss": 0.14849199056625367,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 0.9244444444444444,
189
+ "grad_norm": 0.10201010853052139,
190
+ "learning_rate": 0.00017529162375823958,
191
+ "loss": 0.12720253467559814,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 0.96,
196
+ "grad_norm": 0.23005172610282898,
197
+ "learning_rate": 0.00017251138303982675,
198
+ "loss": 0.11229866743087769,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 0.9955555555555555,
203
+ "grad_norm": 0.15584969520568848,
204
+ "learning_rate": 0.00016960758316310597,
205
+ "loss": 0.12440342903137207,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 1.0284444444444445,
210
+ "grad_norm": 0.08643736690282822,
211
+ "learning_rate": 0.0001665851721935205,
212
+ "loss": 0.0838462769985199,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 1.064,
217
+ "grad_norm": 0.061173878610134125,
218
+ "learning_rate": 0.0001634493003094259,
219
+ "loss": 0.05495935678482056,
220
+ "step": 300
221
+ },
222
+ {
223
+ "epoch": 1.0995555555555556,
224
+ "grad_norm": 0.3276287019252777,
225
+ "learning_rate": 0.00016020531102620304,
226
+ "loss": 0.055119764804840085,
227
+ "step": 310
228
+ },
229
+ {
230
+ "epoch": 1.1351111111111112,
231
+ "grad_norm": 0.1756429374217987,
232
+ "learning_rate": 0.0001568587320909255,
233
+ "loss": 0.0650447130203247,
234
+ "step": 320
235
+ },
236
+ {
237
+ "epoch": 1.1706666666666667,
238
+ "grad_norm": 0.1355263888835907,
239
+ "learning_rate": 0.00015341526606309645,
240
+ "loss": 0.06544734239578247,
241
+ "step": 330
242
+ },
243
+ {
244
+ "epoch": 1.2062222222222223,
245
+ "grad_norm": 0.20916008949279785,
246
+ "learning_rate": 0.00014988078059750652,
247
+ "loss": 0.06727538704872131,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 1.2417777777777779,
252
+ "grad_norm": 0.19566652178764343,
253
+ "learning_rate": 0.00014626129844576893,
254
+ "loss": 0.06408223509788513,
255
+ "step": 350
256
+ },
257
+ {
258
+ "epoch": 1.2773333333333334,
259
+ "grad_norm": 0.05987081304192543,
260
+ "learning_rate": 0.00014256298719357062,
261
+ "loss": 0.04535002112388611,
262
+ "step": 360
263
+ },
264
+ {
265
+ "epoch": 1.3128888888888888,
266
+ "grad_norm": 0.1902439296245575,
267
+ "learning_rate": 0.00013879214875112665,
268
+ "loss": 0.04922315180301666,
269
+ "step": 370
270
+ },
271
+ {
272
+ "epoch": 1.3484444444444446,
273
+ "grad_norm": 0.25207415223121643,
274
+ "learning_rate": 0.00013495520861474565,
275
+ "loss": 0.060137057304382326,
276
+ "step": 380
277
+ },
278
+ {
279
+ "epoch": 1.384,
280
+ "grad_norm": 0.34174537658691406,
281
+ "learning_rate": 0.00013105870491780558,
282
+ "loss": 0.05463656783103943,
283
+ "step": 390
284
+ },
285
+ {
286
+ "epoch": 1.4195555555555557,
287
+ "grad_norm": 0.19624291360378265,
288
+ "learning_rate": 0.00012710927728979568,
289
+ "loss": 0.06046912670135498,
290
+ "step": 400
291
+ },
292
+ {
293
+ "epoch": 1.455111111111111,
294
+ "grad_norm": 0.22849531471729279,
295
+ "learning_rate": 0.00012311365554240971,
296
+ "loss": 0.039173880219459535,
297
+ "step": 410
298
+ },
299
+ {
300
+ "epoch": 1.4906666666666666,
301
+ "grad_norm": 0.22941145300865173,
302
+ "learning_rate": 0.0001190786482019691,
303
+ "loss": 0.04884783029556274,
304
+ "step": 420
305
+ },
306
+ {
307
+ "epoch": 1.5262222222222221,
308
+ "grad_norm": 0.2867370843887329,
309
+ "learning_rate": 0.00011501113090771619,
310
+ "loss": 0.05086652636528015,
311
+ "step": 430
312
+ },
313
+ {
314
+ "epoch": 1.561777777777778,
315
+ "grad_norm": 0.27643826603889465,
316
+ "learning_rate": 0.00011091803469574789,
317
+ "loss": 0.04540249407291412,
318
+ "step": 440
319
+ },
320
+ {
321
+ "epoch": 1.5973333333333333,
322
+ "grad_norm": 0.1365540772676468,
323
+ "learning_rate": 0.00010680633418855267,
324
+ "loss": 0.0644813358783722,
325
+ "step": 450
326
+ },
327
+ {
328
+ "epoch": 1.6328888888888888,
329
+ "grad_norm": 0.21614767611026764,
330
+ "learning_rate": 0.00010268303571027696,
331
+ "loss": 0.07455622553825378,
332
+ "step": 460
333
+ },
334
+ {
335
+ "epoch": 1.6684444444444444,
336
+ "grad_norm": 0.08498027920722961,
337
+ "learning_rate": 9.855516534797187e-05,
338
+ "loss": 0.04775569438934326,
339
+ "step": 470
340
+ },
341
+ {
342
+ "epoch": 1.704,
343
+ "grad_norm": 0.2029683142900467,
344
+ "learning_rate": 9.442975697916372e-05,
345
+ "loss": 0.04402676820755005,
346
+ "step": 480
347
+ },
348
+ {
349
+ "epoch": 1.7395555555555555,
350
+ "grad_norm": 0.34462177753448486,
351
+ "learning_rate": 9.031384028615004e-05,
352
+ "loss": 0.05336908102035522,
353
+ "step": 490
354
+ },
355
+ {
356
+ "epoch": 1.775111111111111,
357
+ "grad_norm": 0.05993572995066643,
358
+ "learning_rate": 8.621442877744409e-05,
359
+ "loss": 0.04580667018890381,
360
+ "step": 500
361
+ },
362
+ {
363
+ "epoch": 1.775111111111111,
364
+ "eval_accuracy": 0.9426666666666668,
365
+ "eval_loss": 0.15560266375541687,
366
+ "eval_mcq_accuracy": 0.7222222222222222,
367
+ "eval_runtime": 26.3716,
368
+ "eval_samples_per_second": 17.064,
369
+ "eval_steps_per_second": 4.285,
370
+ "step": 500
371
+ }
372
+ ],
373
+ "logging_steps": 10,
374
+ "max_steps": 846,
375
+ "num_input_tokens_seen": 0,
376
+ "num_train_epochs": 3,
377
+ "save_steps": 500,
378
+ "stateful_callbacks": {
379
+ "TrainerControl": {
380
+ "args": {
381
+ "should_epoch_stop": false,
382
+ "should_evaluate": false,
383
+ "should_log": false,
384
+ "should_save": true,
385
+ "should_training_stop": false
386
+ },
387
+ "attributes": {}
388
+ }
389
+ },
390
+ "total_flos": 6.465519316726579e+16,
391
+ "train_batch_size": 4,
392
+ "trial_name": null,
393
+ "trial_params": null
394
+ }
checkpoint-500/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06e72e4c171697e89ed577da3f438cb2d6210bd6fe361313102cda2f6b7351b8
3
+ size 5649
checkpoint-846/README.md ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: google/gemma-2-9b-it
3
+ library_name: peft
4
+ pipeline_tag: text-generation
5
+ tags:
6
+ - base_model:adapter:google/gemma-2-9b-it
7
+ - llama-factory
8
+ - lora
9
+ - transformers
10
+ ---
11
+
12
+ # Model Card for Model ID
13
+
14
+ <!-- Provide a quick summary of what the model is/does. -->
15
+
16
+
17
+
18
+ ## Model Details
19
+
20
+ ### Model Description
21
+
22
+ <!-- Provide a longer summary of what this model is. -->
23
+
24
+
25
+
26
+ - **Developed by:** [More Information Needed]
27
+ - **Funded by [optional]:** [More Information Needed]
28
+ - **Shared by [optional]:** [More Information Needed]
29
+ - **Model type:** [More Information Needed]
30
+ - **Language(s) (NLP):** [More Information Needed]
31
+ - **License:** [More Information Needed]
32
+ - **Finetuned from model [optional]:** [More Information Needed]
33
+
34
+ ### Model Sources [optional]
35
+
36
+ <!-- Provide the basic links for the model. -->
37
+
38
+ - **Repository:** [More Information Needed]
39
+ - **Paper [optional]:** [More Information Needed]
40
+ - **Demo [optional]:** [More Information Needed]
41
+
42
+ ## Uses
43
+
44
+ <!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
45
+
46
+ ### Direct Use
47
+
48
+ <!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
49
+
50
+ [More Information Needed]
51
+
52
+ ### Downstream Use [optional]
53
+
54
+ <!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
55
+
56
+ [More Information Needed]
57
+
58
+ ### Out-of-Scope Use
59
+
60
+ <!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
61
+
62
+ [More Information Needed]
63
+
64
+ ## Bias, Risks, and Limitations
65
+
66
+ <!-- This section is meant to convey both technical and sociotechnical limitations. -->
67
+
68
+ [More Information Needed]
69
+
70
+ ### Recommendations
71
+
72
+ <!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
73
+
74
+ Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
75
+
76
+ ## How to Get Started with the Model
77
+
78
+ Use the code below to get started with the model.
79
+
80
+ [More Information Needed]
81
+
82
+ ## Training Details
83
+
84
+ ### Training Data
85
+
86
+ <!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
87
+
88
+ [More Information Needed]
89
+
90
+ ### Training Procedure
91
+
92
+ <!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
93
+
94
+ #### Preprocessing [optional]
95
+
96
+ [More Information Needed]
97
+
98
+
99
+ #### Training Hyperparameters
100
+
101
+ - **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
102
+
103
+ #### Speeds, Sizes, Times [optional]
104
+
105
+ <!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
106
+
107
+ [More Information Needed]
108
+
109
+ ## Evaluation
110
+
111
+ <!-- This section describes the evaluation protocols and provides the results. -->
112
+
113
+ ### Testing Data, Factors & Metrics
114
+
115
+ #### Testing Data
116
+
117
+ <!-- This should link to a Dataset Card if possible. -->
118
+
119
+ [More Information Needed]
120
+
121
+ #### Factors
122
+
123
+ <!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
124
+
125
+ [More Information Needed]
126
+
127
+ #### Metrics
128
+
129
+ <!-- These are the evaluation metrics being used, ideally with a description of why. -->
130
+
131
+ [More Information Needed]
132
+
133
+ ### Results
134
+
135
+ [More Information Needed]
136
+
137
+ #### Summary
138
+
139
+
140
+
141
+ ## Model Examination [optional]
142
+
143
+ <!-- Relevant interpretability work for the model goes here -->
144
+
145
+ [More Information Needed]
146
+
147
+ ## Environmental Impact
148
+
149
+ <!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
150
+
151
+ Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
152
+
153
+ - **Hardware Type:** [More Information Needed]
154
+ - **Hours used:** [More Information Needed]
155
+ - **Cloud Provider:** [More Information Needed]
156
+ - **Compute Region:** [More Information Needed]
157
+ - **Carbon Emitted:** [More Information Needed]
158
+
159
+ ## Technical Specifications [optional]
160
+
161
+ ### Model Architecture and Objective
162
+
163
+ [More Information Needed]
164
+
165
+ ### Compute Infrastructure
166
+
167
+ [More Information Needed]
168
+
169
+ #### Hardware
170
+
171
+ [More Information Needed]
172
+
173
+ #### Software
174
+
175
+ [More Information Needed]
176
+
177
+ ## Citation [optional]
178
+
179
+ <!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
180
+
181
+ **BibTeX:**
182
+
183
+ [More Information Needed]
184
+
185
+ **APA:**
186
+
187
+ [More Information Needed]
188
+
189
+ ## Glossary [optional]
190
+
191
+ <!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
192
+
193
+ [More Information Needed]
194
+
195
+ ## More Information [optional]
196
+
197
+ [More Information Needed]
198
+
199
+ ## Model Card Authors [optional]
200
+
201
+ [More Information Needed]
202
+
203
+ ## Model Card Contact
204
+
205
+ [More Information Needed]
206
+ ### Framework versions
207
+
208
+ - PEFT 0.18.1
checkpoint-846/adapter_config.json ADDED
@@ -0,0 +1,46 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "google/gemma-2-9b-it",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 32,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.05,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 16,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "down_proj",
34
+ "o_proj",
35
+ "k_proj",
36
+ "up_proj",
37
+ "v_proj",
38
+ "q_proj"
39
+ ],
40
+ "target_parameters": null,
41
+ "task_type": "CAUSAL_LM",
42
+ "trainable_token_indices": null,
43
+ "use_dora": false,
44
+ "use_qalora": false,
45
+ "use_rslora": false
46
+ }
checkpoint-846/adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:25d2961ba873b47b3a0145266448f754de82479dc07e1e2a42d8e25899ec8820
3
+ size 216151256
checkpoint-846/chat_template.jinja ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {{ bos_token }}{% if messages[0]['role'] == 'system' %}{{ raise_exception('System role not supported') }}{% endif %}{% for message in messages %}{% if (message['role'] == 'user') != (loop.index0 % 2 == 0) %}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{% endif %}{% if (message['role'] == 'assistant') %}{% set role = 'model' %}{% else %}{% set role = message['role'] %}{% endif %}{{ '<start_of_turn>' + role + '
2
+ ' + message['content'] | trim + '<end_of_turn>
3
+ ' }}{% endfor %}{% if add_generation_prompt %}{{'<start_of_turn>model
4
+ '}}{% endif %}
checkpoint-846/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:394ace002a144ac6ad5486387502f2d36f70c087310c3d907857240c76fcb36e
3
+ size 34362748
checkpoint-846/tokenizer_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "bos_token": "<bos>",
4
+ "clean_up_tokenization_spaces": false,
5
+ "eos_token": "<eos>",
6
+ "extra_special_tokens": [
7
+ "<eos>",
8
+ "<end_of_turn>"
9
+ ],
10
+ "is_local": false,
11
+ "mask_token": "<mask>",
12
+ "model_max_length": 1000000000000000019884624838656,
13
+ "pad_token": "<pad>",
14
+ "padding_side": "right",
15
+ "sp_model_kwargs": {},
16
+ "spaces_between_special_tokens": false,
17
+ "split_special_tokens": false,
18
+ "tokenizer_class": "GemmaTokenizer",
19
+ "unk_token": "<unk>",
20
+ "use_default_system_prompt": false
21
+ }
checkpoint-846/trainer_state.json ADDED
@@ -0,0 +1,632 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 3.0,
6
+ "eval_steps": 500,
7
+ "global_step": 846,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.035555555555555556,
14
+ "grad_norm": 0.7644901871681213,
15
+ "learning_rate": 2.1176470588235296e-05,
16
+ "loss": 0.30520846843719485,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.07111111111111111,
21
+ "grad_norm": 0.43560856580734253,
22
+ "learning_rate": 4.470588235294118e-05,
23
+ "loss": 0.26438264846801757,
24
+ "step": 20
25
+ },
26
+ {
27
+ "epoch": 0.10666666666666667,
28
+ "grad_norm": 0.2563800811767578,
29
+ "learning_rate": 6.823529411764707e-05,
30
+ "loss": 0.19181081056594848,
31
+ "step": 30
32
+ },
33
+ {
34
+ "epoch": 0.14222222222222222,
35
+ "grad_norm": 0.10259224474430084,
36
+ "learning_rate": 9.176470588235295e-05,
37
+ "loss": 0.15193926095962523,
38
+ "step": 40
39
+ },
40
+ {
41
+ "epoch": 0.17777777777777778,
42
+ "grad_norm": 0.1993834525346756,
43
+ "learning_rate": 0.00011529411764705881,
44
+ "loss": 0.13089948892593384,
45
+ "step": 50
46
+ },
47
+ {
48
+ "epoch": 0.21333333333333335,
49
+ "grad_norm": 0.19039888679981232,
50
+ "learning_rate": 0.00013882352941176472,
51
+ "loss": 0.1438336730003357,
52
+ "step": 60
53
+ },
54
+ {
55
+ "epoch": 0.24888888888888888,
56
+ "grad_norm": 0.153082013130188,
57
+ "learning_rate": 0.0001623529411764706,
58
+ "loss": 0.13305349349975587,
59
+ "step": 70
60
+ },
61
+ {
62
+ "epoch": 0.28444444444444444,
63
+ "grad_norm": 0.08913639187812805,
64
+ "learning_rate": 0.00018588235294117648,
65
+ "loss": 0.16499919891357423,
66
+ "step": 80
67
+ },
68
+ {
69
+ "epoch": 0.32,
70
+ "grad_norm": 0.18134109675884247,
71
+ "learning_rate": 0.00019998636639992777,
72
+ "loss": 0.13644593954086304,
73
+ "step": 90
74
+ },
75
+ {
76
+ "epoch": 0.35555555555555557,
77
+ "grad_norm": 0.09949612617492676,
78
+ "learning_rate": 0.00019983303108908946,
79
+ "loss": 0.13039498329162597,
80
+ "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.39111111111111113,
84
+ "grad_norm": 0.12579314410686493,
85
+ "learning_rate": 0.00019950958062149127,
86
+ "loss": 0.1297929048538208,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.4266666666666667,
91
+ "grad_norm": 0.1197328120470047,
92
+ "learning_rate": 0.00019901656615566656,
93
+ "loss": 0.1509210705757141,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.4622222222222222,
98
+ "grad_norm": 0.15709449350833893,
99
+ "learning_rate": 0.00019835482778664425,
100
+ "loss": 0.14251822233200073,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.49777777777777776,
105
+ "grad_norm": 0.2203192412853241,
106
+ "learning_rate": 0.0001975254931144296,
107
+ "loss": 0.1412465214729309,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 0.5333333333333333,
112
+ "grad_norm": 0.32334810495376587,
113
+ "learning_rate": 0.0001965299753225775,
114
+ "loss": 0.1519417643547058,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 0.5688888888888889,
119
+ "grad_norm": 0.1700880080461502,
120
+ "learning_rate": 0.00019536997077013236,
121
+ "loss": 0.13457289934158326,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 0.6044444444444445,
126
+ "grad_norm": 0.17782148718833923,
127
+ "learning_rate": 0.00019404745610103786,
128
+ "loss": 0.14420714378356933,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 0.64,
133
+ "grad_norm": 0.19534672796726227,
134
+ "learning_rate": 0.00019256468487594214,
135
+ "loss": 0.1206929087638855,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 0.6755555555555556,
140
+ "grad_norm": 0.12961238622665405,
141
+ "learning_rate": 0.00019092418373213796,
142
+ "loss": 0.13416168689727784,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 0.7111111111111111,
147
+ "grad_norm": 0.19645075500011444,
148
+ "learning_rate": 0.000189128748078181,
149
+ "loss": 0.13229730129241943,
150
+ "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.7466666666666667,
154
+ "grad_norm": 0.17324262857437134,
155
+ "learning_rate": 0.00018718143733052278,
156
+ "loss": 0.12280937433242797,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.7822222222222223,
161
+ "grad_norm": 0.10764078050851822,
162
+ "learning_rate": 0.0001850855697002753,
163
+ "loss": 0.12583138942718505,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.8177777777777778,
168
+ "grad_norm": 0.13804763555526733,
169
+ "learning_rate": 0.00018284471653898994,
170
+ "loss": 0.13474913835525512,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.8533333333333334,
175
+ "grad_norm": 0.11816427856683731,
176
+ "learning_rate": 0.00018046269625308648,
177
+ "loss": 0.13484352827072144,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 0.8888888888888888,
182
+ "grad_norm": 0.09677577018737793,
183
+ "learning_rate": 0.00017794356779730084,
184
+ "loss": 0.14849199056625367,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 0.9244444444444444,
189
+ "grad_norm": 0.10201010853052139,
190
+ "learning_rate": 0.00017529162375823958,
191
+ "loss": 0.12720253467559814,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 0.96,
196
+ "grad_norm": 0.23005172610282898,
197
+ "learning_rate": 0.00017251138303982675,
198
+ "loss": 0.11229866743087769,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 0.9955555555555555,
203
+ "grad_norm": 0.15584969520568848,
204
+ "learning_rate": 0.00016960758316310597,
205
+ "loss": 0.12440342903137207,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 1.0284444444444445,
210
+ "grad_norm": 0.08643736690282822,
211
+ "learning_rate": 0.0001665851721935205,
212
+ "loss": 0.0838462769985199,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 1.064,
217
+ "grad_norm": 0.061173878610134125,
218
+ "learning_rate": 0.0001634493003094259,
219
+ "loss": 0.05495935678482056,
220
+ "step": 300
221
+ },
222
+ {
223
+ "epoch": 1.0995555555555556,
224
+ "grad_norm": 0.3276287019252777,
225
+ "learning_rate": 0.00016020531102620304,
226
+ "loss": 0.055119764804840085,
227
+ "step": 310
228
+ },
229
+ {
230
+ "epoch": 1.1351111111111112,
231
+ "grad_norm": 0.1756429374217987,
232
+ "learning_rate": 0.0001568587320909255,
233
+ "loss": 0.0650447130203247,
234
+ "step": 320
235
+ },
236
+ {
237
+ "epoch": 1.1706666666666667,
238
+ "grad_norm": 0.1355263888835907,
239
+ "learning_rate": 0.00015341526606309645,
240
+ "loss": 0.06544734239578247,
241
+ "step": 330
242
+ },
243
+ {
244
+ "epoch": 1.2062222222222223,
245
+ "grad_norm": 0.20916008949279785,
246
+ "learning_rate": 0.00014988078059750652,
247
+ "loss": 0.06727538704872131,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 1.2417777777777779,
252
+ "grad_norm": 0.19566652178764343,
253
+ "learning_rate": 0.00014626129844576893,
254
+ "loss": 0.06408223509788513,
255
+ "step": 350
256
+ },
257
+ {
258
+ "epoch": 1.2773333333333334,
259
+ "grad_norm": 0.05987081304192543,
260
+ "learning_rate": 0.00014256298719357062,
261
+ "loss": 0.04535002112388611,
262
+ "step": 360
263
+ },
264
+ {
265
+ "epoch": 1.3128888888888888,
266
+ "grad_norm": 0.1902439296245575,
267
+ "learning_rate": 0.00013879214875112665,
268
+ "loss": 0.04922315180301666,
269
+ "step": 370
270
+ },
271
+ {
272
+ "epoch": 1.3484444444444446,
273
+ "grad_norm": 0.25207415223121643,
274
+ "learning_rate": 0.00013495520861474565,
275
+ "loss": 0.060137057304382326,
276
+ "step": 380
277
+ },
278
+ {
279
+ "epoch": 1.384,
280
+ "grad_norm": 0.34174537658691406,
281
+ "learning_rate": 0.00013105870491780558,
282
+ "loss": 0.05463656783103943,
283
+ "step": 390
284
+ },
285
+ {
286
+ "epoch": 1.4195555555555557,
287
+ "grad_norm": 0.19624291360378265,
288
+ "learning_rate": 0.00012710927728979568,
289
+ "loss": 0.06046912670135498,
290
+ "step": 400
291
+ },
292
+ {
293
+ "epoch": 1.455111111111111,
294
+ "grad_norm": 0.22849531471729279,
295
+ "learning_rate": 0.00012311365554240971,
296
+ "loss": 0.039173880219459535,
297
+ "step": 410
298
+ },
299
+ {
300
+ "epoch": 1.4906666666666666,
301
+ "grad_norm": 0.22941145300865173,
302
+ "learning_rate": 0.0001190786482019691,
303
+ "loss": 0.04884783029556274,
304
+ "step": 420
305
+ },
306
+ {
307
+ "epoch": 1.5262222222222221,
308
+ "grad_norm": 0.2867370843887329,
309
+ "learning_rate": 0.00011501113090771619,
310
+ "loss": 0.05086652636528015,
311
+ "step": 430
312
+ },
313
+ {
314
+ "epoch": 1.561777777777778,
315
+ "grad_norm": 0.27643826603889465,
316
+ "learning_rate": 0.00011091803469574789,
317
+ "loss": 0.04540249407291412,
318
+ "step": 440
319
+ },
320
+ {
321
+ "epoch": 1.5973333333333333,
322
+ "grad_norm": 0.1365540772676468,
323
+ "learning_rate": 0.00010680633418855267,
324
+ "loss": 0.0644813358783722,
325
+ "step": 450
326
+ },
327
+ {
328
+ "epoch": 1.6328888888888888,
329
+ "grad_norm": 0.21614767611026764,
330
+ "learning_rate": 0.00010268303571027696,
331
+ "loss": 0.07455622553825378,
332
+ "step": 460
333
+ },
334
+ {
335
+ "epoch": 1.6684444444444444,
336
+ "grad_norm": 0.08498027920722961,
337
+ "learning_rate": 9.855516534797187e-05,
338
+ "loss": 0.04775569438934326,
339
+ "step": 470
340
+ },
341
+ {
342
+ "epoch": 1.704,
343
+ "grad_norm": 0.2029683142900467,
344
+ "learning_rate": 9.442975697916372e-05,
345
+ "loss": 0.04402676820755005,
346
+ "step": 480
347
+ },
348
+ {
349
+ "epoch": 1.7395555555555555,
350
+ "grad_norm": 0.34462177753448486,
351
+ "learning_rate": 9.031384028615004e-05,
352
+ "loss": 0.05336908102035522,
353
+ "step": 490
354
+ },
355
+ {
356
+ "epoch": 1.775111111111111,
357
+ "grad_norm": 0.05993572995066643,
358
+ "learning_rate": 8.621442877744409e-05,
359
+ "loss": 0.04580667018890381,
360
+ "step": 500
361
+ },
362
+ {
363
+ "epoch": 1.775111111111111,
364
+ "eval_accuracy": 0.9426666666666668,
365
+ "eval_loss": 0.15560266375541687,
366
+ "eval_mcq_accuracy": 0.7222222222222222,
367
+ "eval_runtime": 26.3716,
368
+ "eval_samples_per_second": 17.064,
369
+ "eval_steps_per_second": 4.285,
370
+ "step": 500
371
+ },
372
+ {
373
+ "epoch": 1.8106666666666666,
374
+ "grad_norm": 0.3891923427581787,
375
+ "learning_rate": 8.213850783677925e-05,
376
+ "loss": 0.03394646048545837,
377
+ "step": 510
378
+ },
379
+ {
380
+ "epoch": 1.8462222222222222,
381
+ "grad_norm": 0.1975124329328537,
382
+ "learning_rate": 7.809302282003823e-05,
383
+ "loss": 0.054677408933639524,
384
+ "step": 520
385
+ },
386
+ {
387
+ "epoch": 1.8817777777777778,
388
+ "grad_norm": 0.1962609589099884,
389
+ "learning_rate": 7.408486722038943e-05,
390
+ "loss": 0.06665679812431335,
391
+ "step": 530
392
+ },
393
+ {
394
+ "epoch": 1.9173333333333333,
395
+ "grad_norm": 0.2969353199005127,
396
+ "learning_rate": 7.012087092179724e-05,
397
+ "loss": 0.048915204405784604,
398
+ "step": 540
399
+ },
400
+ {
401
+ "epoch": 1.952888888888889,
402
+ "grad_norm": 0.23223088681697845,
403
+ "learning_rate": 6.620778856092227e-05,
404
+ "loss": 0.04671376347541809,
405
+ "step": 550
406
+ },
407
+ {
408
+ "epoch": 1.9884444444444445,
409
+ "grad_norm": 0.4542539417743683,
410
+ "learning_rate": 6.235228801724253e-05,
411
+ "loss": 0.05778223276138306,
412
+ "step": 560
413
+ },
414
+ {
415
+ "epoch": 2.021333333333333,
416
+ "grad_norm": 0.09194125235080719,
417
+ "learning_rate": 5.856093905100899e-05,
418
+ "loss": 0.022845838963985444,
419
+ "step": 570
420
+ },
421
+ {
422
+ "epoch": 2.056888888888889,
423
+ "grad_norm": 0.10670984536409378,
424
+ "learning_rate": 5.4840202108395466e-05,
425
+ "loss": 0.00663934126496315,
426
+ "step": 580
427
+ },
428
+ {
429
+ "epoch": 2.0924444444444443,
430
+ "grad_norm": 0.015582868829369545,
431
+ "learning_rate": 5.119641731291971e-05,
432
+ "loss": 0.007281148433685302,
433
+ "step": 590
434
+ },
435
+ {
436
+ "epoch": 2.128,
437
+ "grad_norm": 0.17645655572414398,
438
+ "learning_rate": 4.7635793661893666e-05,
439
+ "loss": 0.00572865828871727,
440
+ "step": 600
441
+ },
442
+ {
443
+ "epoch": 2.1635555555555555,
444
+ "grad_norm": 0.46430906653404236,
445
+ "learning_rate": 4.416439844631271e-05,
446
+ "loss": 0.008638855814933778,
447
+ "step": 610
448
+ },
449
+ {
450
+ "epoch": 2.1991111111111112,
451
+ "grad_norm": 0.011537984013557434,
452
+ "learning_rate": 4.078814691221139e-05,
453
+ "loss": 0.004736468568444252,
454
+ "step": 620
455
+ },
456
+ {
457
+ "epoch": 2.2346666666666666,
458
+ "grad_norm": 0.00896433461457491,
459
+ "learning_rate": 3.751279218110387e-05,
460
+ "loss": 0.007740923762321472,
461
+ "step": 630
462
+ },
463
+ {
464
+ "epoch": 2.2702222222222224,
465
+ "grad_norm": 0.002848990960046649,
466
+ "learning_rate": 3.434391544668383e-05,
467
+ "loss": 0.007571302354335785,
468
+ "step": 640
469
+ },
470
+ {
471
+ "epoch": 2.3057777777777777,
472
+ "grad_norm": 0.01570839062333107,
473
+ "learning_rate": 3.1286916464488505e-05,
474
+ "loss": 0.006724107265472412,
475
+ "step": 650
476
+ },
477
+ {
478
+ "epoch": 2.3413333333333335,
479
+ "grad_norm": 0.003965158946812153,
480
+ "learning_rate": 2.8347004350733185e-05,
481
+ "loss": 0.008114227652549743,
482
+ "step": 660
483
+ },
484
+ {
485
+ "epoch": 2.376888888888889,
486
+ "grad_norm": 0.10036994516849518,
487
+ "learning_rate": 2.55291887059944e-05,
488
+ "loss": 0.0027894463390111925,
489
+ "step": 670
490
+ },
491
+ {
492
+ "epoch": 2.4124444444444446,
493
+ "grad_norm": 0.1702636480331421,
494
+ "learning_rate": 2.2838271078866714e-05,
495
+ "loss": 0.0071901664137840274,
496
+ "step": 680
497
+ },
498
+ {
499
+ "epoch": 2.448,
500
+ "grad_norm": 0.003337263595312834,
501
+ "learning_rate": 2.0278836784140044e-05,
502
+ "loss": 0.0033892091363668443,
503
+ "step": 690
504
+ },
505
+ {
506
+ "epoch": 2.4835555555555557,
507
+ "grad_norm": 0.0054590729996562,
508
+ "learning_rate": 1.785524708943802e-05,
509
+ "loss": 0.002477823756635189,
510
+ "step": 700
511
+ },
512
+ {
513
+ "epoch": 2.519111111111111,
514
+ "grad_norm": 0.02814805507659912,
515
+ "learning_rate": 1.557163178363251e-05,
516
+ "loss": 0.0032980531454086305,
517
+ "step": 710
518
+ },
519
+ {
520
+ "epoch": 2.554666666666667,
521
+ "grad_norm": 0.01721755601465702,
522
+ "learning_rate": 1.3431882139696916e-05,
523
+ "loss": 0.00092921182513237,
524
+ "step": 720
525
+ },
526
+ {
527
+ "epoch": 2.590222222222222,
528
+ "grad_norm": 0.004700134973973036,
529
+ "learning_rate": 1.1439644283989747e-05,
530
+ "loss": 0.008682060241699218,
531
+ "step": 730
532
+ },
533
+ {
534
+ "epoch": 2.6257777777777775,
535
+ "grad_norm": 0.02318074181675911,
536
+ "learning_rate": 9.59831298326731e-06,
537
+ "loss": 0.0035014577209949494,
538
+ "step": 740
539
+ },
540
+ {
541
+ "epoch": 2.6613333333333333,
542
+ "grad_norm": 0.10873377323150635,
543
+ "learning_rate": 7.911025860012444e-06,
544
+ "loss": 0.007874777913093567,
545
+ "step": 750
546
+ },
547
+ {
548
+ "epoch": 2.696888888888889,
549
+ "grad_norm": 0.007307600695639849,
550
+ "learning_rate": 6.38065804593595e-06,
551
+ "loss": 0.003820793330669403,
552
+ "step": 760
553
+ },
554
+ {
555
+ "epoch": 2.7324444444444445,
556
+ "grad_norm": 0.002649878617376089,
557
+ "learning_rate": 5.009817282761675e-06,
558
+ "loss": 0.0014227939769625663,
559
+ "step": 770
560
+ },
561
+ {
562
+ "epoch": 2.768,
563
+ "grad_norm": 0.02097943052649498,
564
+ "learning_rate": 3.800839478643259e-06,
565
+ "loss": 0.007305952161550522,
566
+ "step": 780
567
+ },
568
+ {
569
+ "epoch": 2.8035555555555556,
570
+ "grad_norm": 0.11966679990291595,
571
+ "learning_rate": 2.7557847277841942e-06,
572
+ "loss": 0.004338302463293075,
573
+ "step": 790
574
+ },
575
+ {
576
+ "epoch": 2.8391111111111114,
577
+ "grad_norm": 0.07167425006628036,
578
+ "learning_rate": 1.8764338000442083e-06,
579
+ "loss": 0.002496876008808613,
580
+ "step": 800
581
+ },
582
+ {
583
+ "epoch": 2.8746666666666667,
584
+ "grad_norm": 0.010830031707882881,
585
+ "learning_rate": 1.1642851065131633e-06,
586
+ "loss": 0.0015433511696755886,
587
+ "step": 810
588
+ },
589
+ {
590
+ "epoch": 2.910222222222222,
591
+ "grad_norm": 0.195694237947464,
592
+ "learning_rate": 6.205521462235186e-07,
593
+ "loss": 0.001970846764743328,
594
+ "step": 820
595
+ },
596
+ {
597
+ "epoch": 2.945777777777778,
598
+ "grad_norm": 0.00489334249868989,
599
+ "learning_rate": 2.4616143835202166e-07,
600
+ "loss": 0.010426017642021179,
601
+ "step": 830
602
+ },
603
+ {
604
+ "epoch": 2.981333333333333,
605
+ "grad_norm": 0.009853039868175983,
606
+ "learning_rate": 4.1750943434026855e-08,
607
+ "loss": 0.0070929393172264096,
608
+ "step": 840
609
+ }
610
+ ],
611
+ "logging_steps": 10,
612
+ "max_steps": 846,
613
+ "num_input_tokens_seen": 0,
614
+ "num_train_epochs": 3,
615
+ "save_steps": 500,
616
+ "stateful_callbacks": {
617
+ "TrainerControl": {
618
+ "args": {
619
+ "should_epoch_stop": false,
620
+ "should_evaluate": false,
621
+ "should_log": false,
622
+ "should_save": true,
623
+ "should_training_stop": true
624
+ },
625
+ "attributes": {}
626
+ }
627
+ },
628
+ "total_flos": 1.0958550940031386e+17,
629
+ "train_batch_size": 4,
630
+ "trial_name": null,
631
+ "trial_params": null
632
+ }
checkpoint-846/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06e72e4c171697e89ed577da3f438cb2d6210bd6fe361313102cda2f6b7351b8
3
+ size 5649
eval_results.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "epoch": 3.0,
3
+ "eval_accuracy": 0.9458148148148149,
4
+ "eval_loss": 0.21385908126831055,
5
+ "eval_mcq_accuracy": 0.7311111111111112,
6
+ "eval_runtime": 26.4009,
7
+ "eval_samples_per_second": 17.045,
8
+ "eval_steps_per_second": 4.28
9
+ }
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:394ace002a144ac6ad5486387502f2d36f70c087310c3d907857240c76fcb36e
3
+ size 34362748
tokenizer_config.json ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "bos_token": "<bos>",
4
+ "clean_up_tokenization_spaces": false,
5
+ "eos_token": "<eos>",
6
+ "extra_special_tokens": [
7
+ "<eos>",
8
+ "<end_of_turn>"
9
+ ],
10
+ "is_local": false,
11
+ "mask_token": "<mask>",
12
+ "model_max_length": 1000000000000000019884624838656,
13
+ "pad_token": "<pad>",
14
+ "padding_side": "right",
15
+ "sp_model_kwargs": {},
16
+ "spaces_between_special_tokens": false,
17
+ "split_special_tokens": false,
18
+ "tokenizer_class": "GemmaTokenizer",
19
+ "unk_token": "<unk>",
20
+ "use_default_system_prompt": false
21
+ }
train_results.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "epoch": 3.0,
3
+ "total_flos": 1.0958550940031386e+17,
4
+ "train_loss": 0.06952556652285546,
5
+ "train_runtime": 2162.4382,
6
+ "train_samples_per_second": 6.243,
7
+ "train_steps_per_second": 0.391
8
+ }
trainer_log.jsonl ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"current_steps": 10, "total_steps": 846, "loss": 0.30520846843719485, "lr": 2.1176470588235296e-05, "epoch": 0.035555555555555556, "percentage": 1.18, "elapsed_time": "0:00:25", "remaining_time": "0:36:12"}
2
+ {"current_steps": 20, "total_steps": 846, "loss": 0.26438264846801757, "lr": 4.470588235294118e-05, "epoch": 0.07111111111111111, "percentage": 2.36, "elapsed_time": "0:00:49", "remaining_time": "0:34:05"}
3
+ {"current_steps": 30, "total_steps": 846, "loss": 0.19181081056594848, "lr": 6.823529411764707e-05, "epoch": 0.10666666666666667, "percentage": 3.55, "elapsed_time": "0:01:14", "remaining_time": "0:33:51"}
4
+ {"current_steps": 40, "total_steps": 846, "loss": 0.15193926095962523, "lr": 9.176470588235295e-05, "epoch": 0.14222222222222222, "percentage": 4.73, "elapsed_time": "0:01:42", "remaining_time": "0:34:33"}
5
+ {"current_steps": 50, "total_steps": 846, "loss": 0.13089948892593384, "lr": 0.00011529411764705881, "epoch": 0.17777777777777778, "percentage": 5.91, "elapsed_time": "0:02:08", "remaining_time": "0:34:01"}
6
+ {"current_steps": 60, "total_steps": 846, "loss": 0.1438336730003357, "lr": 0.00013882352941176472, "epoch": 0.21333333333333335, "percentage": 7.09, "elapsed_time": "0:02:33", "remaining_time": "0:33:30"}
7
+ {"current_steps": 70, "total_steps": 846, "loss": 0.13305349349975587, "lr": 0.0001623529411764706, "epoch": 0.24888888888888888, "percentage": 8.27, "elapsed_time": "0:02:57", "remaining_time": "0:32:46"}
8
+ {"current_steps": 80, "total_steps": 846, "loss": 0.16499919891357423, "lr": 0.00018588235294117648, "epoch": 0.28444444444444444, "percentage": 9.46, "elapsed_time": "0:03:22", "remaining_time": "0:32:16"}
9
+ {"current_steps": 90, "total_steps": 846, "loss": 0.13644593954086304, "lr": 0.00019998636639992777, "epoch": 0.32, "percentage": 10.64, "elapsed_time": "0:03:46", "remaining_time": "0:31:39"}
10
+ {"current_steps": 100, "total_steps": 846, "loss": 0.13039498329162597, "lr": 0.00019983303108908946, "epoch": 0.35555555555555557, "percentage": 11.82, "elapsed_time": "0:04:10", "remaining_time": "0:31:10"}
11
+ {"current_steps": 110, "total_steps": 846, "loss": 0.1297929048538208, "lr": 0.00019950958062149127, "epoch": 0.39111111111111113, "percentage": 13.0, "elapsed_time": "0:04:34", "remaining_time": "0:30:37"}
12
+ {"current_steps": 120, "total_steps": 846, "loss": 0.1509210705757141, "lr": 0.00019901656615566656, "epoch": 0.4266666666666667, "percentage": 14.18, "elapsed_time": "0:04:59", "remaining_time": "0:30:10"}
13
+ {"current_steps": 130, "total_steps": 846, "loss": 0.14251822233200073, "lr": 0.00019835482778664425, "epoch": 0.4622222222222222, "percentage": 15.37, "elapsed_time": "0:05:25", "remaining_time": "0:29:55"}
14
+ {"current_steps": 140, "total_steps": 846, "loss": 0.1412465214729309, "lr": 0.0001975254931144296, "epoch": 0.49777777777777776, "percentage": 16.55, "elapsed_time": "0:05:52", "remaining_time": "0:29:37"}
15
+ {"current_steps": 150, "total_steps": 846, "loss": 0.1519417643547058, "lr": 0.0001965299753225775, "epoch": 0.5333333333333333, "percentage": 17.73, "elapsed_time": "0:06:15", "remaining_time": "0:29:04"}
16
+ {"current_steps": 160, "total_steps": 846, "loss": 0.13457289934158326, "lr": 0.00019536997077013236, "epoch": 0.5688888888888889, "percentage": 18.91, "elapsed_time": "0:06:41", "remaining_time": "0:28:41"}
17
+ {"current_steps": 170, "total_steps": 846, "loss": 0.14420714378356933, "lr": 0.00019404745610103786, "epoch": 0.6044444444444445, "percentage": 20.09, "elapsed_time": "0:07:09", "remaining_time": "0:28:26"}
18
+ {"current_steps": 180, "total_steps": 846, "loss": 0.1206929087638855, "lr": 0.00019256468487594214, "epoch": 0.64, "percentage": 21.28, "elapsed_time": "0:07:35", "remaining_time": "0:28:04"}
19
+ {"current_steps": 190, "total_steps": 846, "loss": 0.13416168689727784, "lr": 0.00019092418373213796, "epoch": 0.6755555555555556, "percentage": 22.46, "elapsed_time": "0:07:59", "remaining_time": "0:27:34"}
20
+ {"current_steps": 200, "total_steps": 846, "loss": 0.13229730129241943, "lr": 0.000189128748078181, "epoch": 0.7111111111111111, "percentage": 23.64, "elapsed_time": "0:08:24", "remaining_time": "0:27:10"}
21
+ {"current_steps": 210, "total_steps": 846, "loss": 0.12280937433242797, "lr": 0.00018718143733052278, "epoch": 0.7466666666666667, "percentage": 24.82, "elapsed_time": "0:08:52", "remaining_time": "0:26:53"}
22
+ {"current_steps": 220, "total_steps": 846, "loss": 0.12583138942718505, "lr": 0.0001850855697002753, "epoch": 0.7822222222222223, "percentage": 26.0, "elapsed_time": "0:09:17", "remaining_time": "0:26:26"}
23
+ {"current_steps": 230, "total_steps": 846, "loss": 0.13474913835525512, "lr": 0.00018284471653898994, "epoch": 0.8177777777777778, "percentage": 27.19, "elapsed_time": "0:09:42", "remaining_time": "0:25:59"}
24
+ {"current_steps": 240, "total_steps": 846, "loss": 0.13484352827072144, "lr": 0.00018046269625308648, "epoch": 0.8533333333333334, "percentage": 28.37, "elapsed_time": "0:10:07", "remaining_time": "0:25:32"}
25
+ {"current_steps": 250, "total_steps": 846, "loss": 0.14849199056625367, "lr": 0.00017794356779730084, "epoch": 0.8888888888888888, "percentage": 29.55, "elapsed_time": "0:10:31", "remaining_time": "0:25:05"}
26
+ {"current_steps": 260, "total_steps": 846, "loss": 0.12720253467559814, "lr": 0.00017529162375823958, "epoch": 0.9244444444444444, "percentage": 30.73, "elapsed_time": "0:10:56", "remaining_time": "0:24:39"}
27
+ {"current_steps": 270, "total_steps": 846, "loss": 0.11229866743087769, "lr": 0.00017251138303982675, "epoch": 0.96, "percentage": 31.91, "elapsed_time": "0:11:20", "remaining_time": "0:24:12"}
28
+ {"current_steps": 280, "total_steps": 846, "loss": 0.12440342903137207, "lr": 0.00016960758316310597, "epoch": 0.9955555555555555, "percentage": 33.1, "elapsed_time": "0:11:45", "remaining_time": "0:23:47"}
29
+ {"current_steps": 290, "total_steps": 846, "loss": 0.0838462769985199, "lr": 0.0001665851721935205, "epoch": 1.0284444444444445, "percentage": 34.28, "elapsed_time": "0:12:10", "remaining_time": "0:23:20"}
30
+ {"current_steps": 300, "total_steps": 846, "loss": 0.05495935678482056, "lr": 0.0001634493003094259, "epoch": 1.064, "percentage": 35.46, "elapsed_time": "0:12:34", "remaining_time": "0:22:52"}
31
+ {"current_steps": 310, "total_steps": 846, "loss": 0.055119764804840085, "lr": 0.00016020531102620304, "epoch": 1.0995555555555556, "percentage": 36.64, "elapsed_time": "0:12:58", "remaining_time": "0:22:25"}
32
+ {"current_steps": 320, "total_steps": 846, "loss": 0.0650447130203247, "lr": 0.0001568587320909255, "epoch": 1.1351111111111112, "percentage": 37.83, "elapsed_time": "0:13:23", "remaining_time": "0:22:01"}
33
+ {"current_steps": 330, "total_steps": 846, "loss": 0.06544734239578247, "lr": 0.00015341526606309645, "epoch": 1.1706666666666667, "percentage": 39.01, "elapsed_time": "0:13:50", "remaining_time": "0:21:38"}
34
+ {"current_steps": 340, "total_steps": 846, "loss": 0.06727538704872131, "lr": 0.00014988078059750652, "epoch": 1.2062222222222223, "percentage": 40.19, "elapsed_time": "0:14:14", "remaining_time": "0:21:12"}
35
+ {"current_steps": 350, "total_steps": 846, "loss": 0.06408223509788513, "lr": 0.00014626129844576893, "epoch": 1.2417777777777779, "percentage": 41.37, "elapsed_time": "0:14:39", "remaining_time": "0:20:46"}
36
+ {"current_steps": 360, "total_steps": 846, "loss": 0.04535002112388611, "lr": 0.00014256298719357062, "epoch": 1.2773333333333334, "percentage": 42.55, "elapsed_time": "0:15:05", "remaining_time": "0:20:22"}
37
+ {"current_steps": 370, "total_steps": 846, "loss": 0.04922315180301666, "lr": 0.00013879214875112665, "epoch": 1.3128888888888888, "percentage": 43.74, "elapsed_time": "0:15:31", "remaining_time": "0:19:57"}
38
+ {"current_steps": 380, "total_steps": 846, "loss": 0.060137057304382326, "lr": 0.00013495520861474565, "epoch": 1.3484444444444446, "percentage": 44.92, "elapsed_time": "0:15:57", "remaining_time": "0:19:33"}
39
+ {"current_steps": 390, "total_steps": 846, "loss": 0.05463656783103943, "lr": 0.00013105870491780558, "epoch": 1.384, "percentage": 46.1, "elapsed_time": "0:16:21", "remaining_time": "0:19:07"}
40
+ {"current_steps": 400, "total_steps": 846, "loss": 0.06046912670135498, "lr": 0.00012710927728979568, "epoch": 1.4195555555555557, "percentage": 47.28, "elapsed_time": "0:16:46", "remaining_time": "0:18:42"}
41
+ {"current_steps": 410, "total_steps": 846, "loss": 0.039173880219459535, "lr": 0.00012311365554240971, "epoch": 1.455111111111111, "percentage": 48.46, "elapsed_time": "0:17:11", "remaining_time": "0:18:16"}
42
+ {"current_steps": 420, "total_steps": 846, "loss": 0.04884783029556274, "lr": 0.0001190786482019691, "epoch": 1.4906666666666666, "percentage": 49.65, "elapsed_time": "0:17:36", "remaining_time": "0:17:51"}
43
+ {"current_steps": 430, "total_steps": 846, "loss": 0.05086652636528015, "lr": 0.00011501113090771619, "epoch": 1.5262222222222221, "percentage": 50.83, "elapsed_time": "0:18:05", "remaining_time": "0:17:29"}
44
+ {"current_steps": 440, "total_steps": 846, "loss": 0.04540249407291412, "lr": 0.00011091803469574789, "epoch": 1.561777777777778, "percentage": 52.01, "elapsed_time": "0:18:30", "remaining_time": "0:17:04"}
45
+ {"current_steps": 450, "total_steps": 846, "loss": 0.0644813358783722, "lr": 0.00010680633418855267, "epoch": 1.5973333333333333, "percentage": 53.19, "elapsed_time": "0:18:52", "remaining_time": "0:16:37"}
46
+ {"current_steps": 460, "total_steps": 846, "loss": 0.07455622553825378, "lr": 0.00010268303571027696, "epoch": 1.6328888888888888, "percentage": 54.37, "elapsed_time": "0:19:18", "remaining_time": "0:16:12"}
47
+ {"current_steps": 470, "total_steps": 846, "loss": 0.04775569438934326, "lr": 9.855516534797187e-05, "epoch": 1.6684444444444444, "percentage": 55.56, "elapsed_time": "0:19:43", "remaining_time": "0:15:47"}
48
+ {"current_steps": 480, "total_steps": 846, "loss": 0.04402676820755005, "lr": 9.442975697916372e-05, "epoch": 1.704, "percentage": 56.74, "elapsed_time": "0:20:10", "remaining_time": "0:15:22"}
49
+ {"current_steps": 490, "total_steps": 846, "loss": 0.05336908102035522, "lr": 9.031384028615004e-05, "epoch": 1.7395555555555555, "percentage": 57.92, "elapsed_time": "0:20:34", "remaining_time": "0:14:57"}
50
+ {"current_steps": 500, "total_steps": 846, "loss": 0.04580667018890381, "lr": 8.621442877744409e-05, "epoch": 1.775111111111111, "percentage": 59.1, "elapsed_time": "0:20:58", "remaining_time": "0:14:31"}
51
+ {"current_steps": 500, "total_steps": 846, "eval_loss": 0.15560266375541687, "epoch": 1.775111111111111, "percentage": 59.1, "elapsed_time": "0:21:25", "remaining_time": "0:14:49"}
52
+ {"current_steps": 510, "total_steps": 846, "loss": 0.03394646048545837, "lr": 8.213850783677925e-05, "epoch": 1.8106666666666666, "percentage": 60.28, "elapsed_time": "0:21:52", "remaining_time": "0:14:24"}
53
+ {"current_steps": 520, "total_steps": 846, "loss": 0.054677408933639524, "lr": 7.809302282003823e-05, "epoch": 1.8462222222222222, "percentage": 61.47, "elapsed_time": "0:22:16", "remaining_time": "0:13:57"}
54
+ {"current_steps": 530, "total_steps": 846, "loss": 0.06665679812431335, "lr": 7.408486722038943e-05, "epoch": 1.8817777777777778, "percentage": 62.65, "elapsed_time": "0:22:41", "remaining_time": "0:13:31"}
55
+ {"current_steps": 540, "total_steps": 846, "loss": 0.048915204405784604, "lr": 7.012087092179724e-05, "epoch": 1.9173333333333333, "percentage": 63.83, "elapsed_time": "0:23:07", "remaining_time": "0:13:06"}
56
+ {"current_steps": 550, "total_steps": 846, "loss": 0.04671376347541809, "lr": 6.620778856092227e-05, "epoch": 1.952888888888889, "percentage": 65.01, "elapsed_time": "0:23:32", "remaining_time": "0:12:40"}
57
+ {"current_steps": 560, "total_steps": 846, "loss": 0.05778223276138306, "lr": 6.235228801724253e-05, "epoch": 1.9884444444444445, "percentage": 66.19, "elapsed_time": "0:23:59", "remaining_time": "0:12:15"}
58
+ {"current_steps": 570, "total_steps": 846, "loss": 0.022845838963985444, "lr": 5.856093905100899e-05, "epoch": 2.021333333333333, "percentage": 67.38, "elapsed_time": "0:24:24", "remaining_time": "0:11:49"}
59
+ {"current_steps": 580, "total_steps": 846, "loss": 0.00663934126496315, "lr": 5.4840202108395466e-05, "epoch": 2.056888888888889, "percentage": 68.56, "elapsed_time": "0:24:49", "remaining_time": "0:11:23"}
60
+ {"current_steps": 590, "total_steps": 846, "loss": 0.007281148433685302, "lr": 5.119641731291971e-05, "epoch": 2.0924444444444443, "percentage": 69.74, "elapsed_time": "0:25:14", "remaining_time": "0:10:57"}
61
+ {"current_steps": 600, "total_steps": 846, "loss": 0.00572865828871727, "lr": 4.7635793661893666e-05, "epoch": 2.128, "percentage": 70.92, "elapsed_time": "0:25:40", "remaining_time": "0:10:31"}
62
+ {"current_steps": 610, "total_steps": 846, "loss": 0.008638855814933778, "lr": 4.416439844631271e-05, "epoch": 2.1635555555555555, "percentage": 72.1, "elapsed_time": "0:26:05", "remaining_time": "0:10:05"}
63
+ {"current_steps": 620, "total_steps": 846, "loss": 0.004736468568444252, "lr": 4.078814691221139e-05, "epoch": 2.1991111111111112, "percentage": 73.29, "elapsed_time": "0:26:30", "remaining_time": "0:09:39"}
64
+ {"current_steps": 630, "total_steps": 846, "loss": 0.007740923762321472, "lr": 3.751279218110387e-05, "epoch": 2.2346666666666666, "percentage": 74.47, "elapsed_time": "0:26:55", "remaining_time": "0:09:14"}
65
+ {"current_steps": 640, "total_steps": 846, "loss": 0.007571302354335785, "lr": 3.434391544668383e-05, "epoch": 2.2702222222222224, "percentage": 75.65, "elapsed_time": "0:27:19", "remaining_time": "0:08:47"}
66
+ {"current_steps": 650, "total_steps": 846, "loss": 0.006724107265472412, "lr": 3.1286916464488505e-05, "epoch": 2.3057777777777777, "percentage": 76.83, "elapsed_time": "0:27:43", "remaining_time": "0:08:21"}
67
+ {"current_steps": 660, "total_steps": 846, "loss": 0.008114227652549743, "lr": 2.8347004350733185e-05, "epoch": 2.3413333333333335, "percentage": 78.01, "elapsed_time": "0:28:12", "remaining_time": "0:07:57"}
68
+ {"current_steps": 670, "total_steps": 846, "loss": 0.0027894463390111925, "lr": 2.55291887059944e-05, "epoch": 2.376888888888889, "percentage": 79.2, "elapsed_time": "0:28:40", "remaining_time": "0:07:31"}
69
+ {"current_steps": 680, "total_steps": 846, "loss": 0.0071901664137840274, "lr": 2.2838271078866714e-05, "epoch": 2.4124444444444446, "percentage": 80.38, "elapsed_time": "0:29:05", "remaining_time": "0:07:06"}
70
+ {"current_steps": 690, "total_steps": 846, "loss": 0.0033892091363668443, "lr": 2.0278836784140044e-05, "epoch": 2.448, "percentage": 81.56, "elapsed_time": "0:29:29", "remaining_time": "0:06:40"}
71
+ {"current_steps": 700, "total_steps": 846, "loss": 0.002477823756635189, "lr": 1.785524708943802e-05, "epoch": 2.4835555555555557, "percentage": 82.74, "elapsed_time": "0:29:55", "remaining_time": "0:06:14"}
72
+ {"current_steps": 710, "total_steps": 846, "loss": 0.0032980531454086305, "lr": 1.557163178363251e-05, "epoch": 2.519111111111111, "percentage": 83.92, "elapsed_time": "0:30:19", "remaining_time": "0:05:48"}
73
+ {"current_steps": 720, "total_steps": 846, "loss": 0.00092921182513237, "lr": 1.3431882139696916e-05, "epoch": 2.554666666666667, "percentage": 85.11, "elapsed_time": "0:30:43", "remaining_time": "0:05:22"}
74
+ {"current_steps": 730, "total_steps": 846, "loss": 0.008682060241699218, "lr": 1.1439644283989747e-05, "epoch": 2.590222222222222, "percentage": 86.29, "elapsed_time": "0:31:10", "remaining_time": "0:04:57"}
75
+ {"current_steps": 740, "total_steps": 846, "loss": 0.0035014577209949494, "lr": 9.59831298326731e-06, "epoch": 2.6257777777777775, "percentage": 87.47, "elapsed_time": "0:31:34", "remaining_time": "0:04:31"}
76
+ {"current_steps": 750, "total_steps": 846, "loss": 0.007874777913093567, "lr": 7.911025860012444e-06, "epoch": 2.6613333333333333, "percentage": 88.65, "elapsed_time": "0:32:00", "remaining_time": "0:04:05"}
77
+ {"current_steps": 760, "total_steps": 846, "loss": 0.003820793330669403, "lr": 6.38065804593595e-06, "epoch": 2.696888888888889, "percentage": 89.83, "elapsed_time": "0:32:24", "remaining_time": "0:03:40"}
78
+ {"current_steps": 770, "total_steps": 846, "loss": 0.0014227939769625663, "lr": 5.009817282761675e-06, "epoch": 2.7324444444444445, "percentage": 91.02, "elapsed_time": "0:32:48", "remaining_time": "0:03:14"}
79
+ {"current_steps": 780, "total_steps": 846, "loss": 0.007305952161550522, "lr": 3.800839478643259e-06, "epoch": 2.768, "percentage": 92.2, "elapsed_time": "0:33:14", "remaining_time": "0:02:48"}
80
+ {"current_steps": 790, "total_steps": 846, "loss": 0.004338302463293075, "lr": 2.7557847277841942e-06, "epoch": 2.8035555555555556, "percentage": 93.38, "elapsed_time": "0:33:39", "remaining_time": "0:02:23"}
81
+ {"current_steps": 800, "total_steps": 846, "loss": 0.002496876008808613, "lr": 1.8764338000442083e-06, "epoch": 2.8391111111111114, "percentage": 94.56, "elapsed_time": "0:34:07", "remaining_time": "0:01:57"}
82
+ {"current_steps": 810, "total_steps": 846, "loss": 0.0015433511696755886, "lr": 1.1642851065131633e-06, "epoch": 2.8746666666666667, "percentage": 95.74, "elapsed_time": "0:34:32", "remaining_time": "0:01:32"}
83
+ {"current_steps": 820, "total_steps": 846, "loss": 0.001970846764743328, "lr": 6.205521462235186e-07, "epoch": 2.910222222222222, "percentage": 96.93, "elapsed_time": "0:34:57", "remaining_time": "0:01:06"}
84
+ {"current_steps": 830, "total_steps": 846, "loss": 0.010426017642021179, "lr": 2.4616143835202166e-07, "epoch": 2.945777777777778, "percentage": 98.11, "elapsed_time": "0:35:21", "remaining_time": "0:00:40"}
85
+ {"current_steps": 840, "total_steps": 846, "loss": 0.0070929393172264096, "lr": 4.1750943434026855e-08, "epoch": 2.981333333333333, "percentage": 99.29, "elapsed_time": "0:35:46", "remaining_time": "0:00:15"}
86
+ {"current_steps": 846, "total_steps": 846, "epoch": 3.0, "percentage": 100.0, "elapsed_time": "0:36:00", "remaining_time": "0:00:00"}
trainer_state.json ADDED
@@ -0,0 +1,641 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 3.0,
6
+ "eval_steps": 500,
7
+ "global_step": 846,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.035555555555555556,
14
+ "grad_norm": 0.7644901871681213,
15
+ "learning_rate": 2.1176470588235296e-05,
16
+ "loss": 0.30520846843719485,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.07111111111111111,
21
+ "grad_norm": 0.43560856580734253,
22
+ "learning_rate": 4.470588235294118e-05,
23
+ "loss": 0.26438264846801757,
24
+ "step": 20
25
+ },
26
+ {
27
+ "epoch": 0.10666666666666667,
28
+ "grad_norm": 0.2563800811767578,
29
+ "learning_rate": 6.823529411764707e-05,
30
+ "loss": 0.19181081056594848,
31
+ "step": 30
32
+ },
33
+ {
34
+ "epoch": 0.14222222222222222,
35
+ "grad_norm": 0.10259224474430084,
36
+ "learning_rate": 9.176470588235295e-05,
37
+ "loss": 0.15193926095962523,
38
+ "step": 40
39
+ },
40
+ {
41
+ "epoch": 0.17777777777777778,
42
+ "grad_norm": 0.1993834525346756,
43
+ "learning_rate": 0.00011529411764705881,
44
+ "loss": 0.13089948892593384,
45
+ "step": 50
46
+ },
47
+ {
48
+ "epoch": 0.21333333333333335,
49
+ "grad_norm": 0.19039888679981232,
50
+ "learning_rate": 0.00013882352941176472,
51
+ "loss": 0.1438336730003357,
52
+ "step": 60
53
+ },
54
+ {
55
+ "epoch": 0.24888888888888888,
56
+ "grad_norm": 0.153082013130188,
57
+ "learning_rate": 0.0001623529411764706,
58
+ "loss": 0.13305349349975587,
59
+ "step": 70
60
+ },
61
+ {
62
+ "epoch": 0.28444444444444444,
63
+ "grad_norm": 0.08913639187812805,
64
+ "learning_rate": 0.00018588235294117648,
65
+ "loss": 0.16499919891357423,
66
+ "step": 80
67
+ },
68
+ {
69
+ "epoch": 0.32,
70
+ "grad_norm": 0.18134109675884247,
71
+ "learning_rate": 0.00019998636639992777,
72
+ "loss": 0.13644593954086304,
73
+ "step": 90
74
+ },
75
+ {
76
+ "epoch": 0.35555555555555557,
77
+ "grad_norm": 0.09949612617492676,
78
+ "learning_rate": 0.00019983303108908946,
79
+ "loss": 0.13039498329162597,
80
+ "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.39111111111111113,
84
+ "grad_norm": 0.12579314410686493,
85
+ "learning_rate": 0.00019950958062149127,
86
+ "loss": 0.1297929048538208,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.4266666666666667,
91
+ "grad_norm": 0.1197328120470047,
92
+ "learning_rate": 0.00019901656615566656,
93
+ "loss": 0.1509210705757141,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.4622222222222222,
98
+ "grad_norm": 0.15709449350833893,
99
+ "learning_rate": 0.00019835482778664425,
100
+ "loss": 0.14251822233200073,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.49777777777777776,
105
+ "grad_norm": 0.2203192412853241,
106
+ "learning_rate": 0.0001975254931144296,
107
+ "loss": 0.1412465214729309,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 0.5333333333333333,
112
+ "grad_norm": 0.32334810495376587,
113
+ "learning_rate": 0.0001965299753225775,
114
+ "loss": 0.1519417643547058,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 0.5688888888888889,
119
+ "grad_norm": 0.1700880080461502,
120
+ "learning_rate": 0.00019536997077013236,
121
+ "loss": 0.13457289934158326,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 0.6044444444444445,
126
+ "grad_norm": 0.17782148718833923,
127
+ "learning_rate": 0.00019404745610103786,
128
+ "loss": 0.14420714378356933,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 0.64,
133
+ "grad_norm": 0.19534672796726227,
134
+ "learning_rate": 0.00019256468487594214,
135
+ "loss": 0.1206929087638855,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 0.6755555555555556,
140
+ "grad_norm": 0.12961238622665405,
141
+ "learning_rate": 0.00019092418373213796,
142
+ "loss": 0.13416168689727784,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 0.7111111111111111,
147
+ "grad_norm": 0.19645075500011444,
148
+ "learning_rate": 0.000189128748078181,
149
+ "loss": 0.13229730129241943,
150
+ "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.7466666666666667,
154
+ "grad_norm": 0.17324262857437134,
155
+ "learning_rate": 0.00018718143733052278,
156
+ "loss": 0.12280937433242797,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.7822222222222223,
161
+ "grad_norm": 0.10764078050851822,
162
+ "learning_rate": 0.0001850855697002753,
163
+ "loss": 0.12583138942718505,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.8177777777777778,
168
+ "grad_norm": 0.13804763555526733,
169
+ "learning_rate": 0.00018284471653898994,
170
+ "loss": 0.13474913835525512,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.8533333333333334,
175
+ "grad_norm": 0.11816427856683731,
176
+ "learning_rate": 0.00018046269625308648,
177
+ "loss": 0.13484352827072144,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 0.8888888888888888,
182
+ "grad_norm": 0.09677577018737793,
183
+ "learning_rate": 0.00017794356779730084,
184
+ "loss": 0.14849199056625367,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 0.9244444444444444,
189
+ "grad_norm": 0.10201010853052139,
190
+ "learning_rate": 0.00017529162375823958,
191
+ "loss": 0.12720253467559814,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 0.96,
196
+ "grad_norm": 0.23005172610282898,
197
+ "learning_rate": 0.00017251138303982675,
198
+ "loss": 0.11229866743087769,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 0.9955555555555555,
203
+ "grad_norm": 0.15584969520568848,
204
+ "learning_rate": 0.00016960758316310597,
205
+ "loss": 0.12440342903137207,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 1.0284444444444445,
210
+ "grad_norm": 0.08643736690282822,
211
+ "learning_rate": 0.0001665851721935205,
212
+ "loss": 0.0838462769985199,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 1.064,
217
+ "grad_norm": 0.061173878610134125,
218
+ "learning_rate": 0.0001634493003094259,
219
+ "loss": 0.05495935678482056,
220
+ "step": 300
221
+ },
222
+ {
223
+ "epoch": 1.0995555555555556,
224
+ "grad_norm": 0.3276287019252777,
225
+ "learning_rate": 0.00016020531102620304,
226
+ "loss": 0.055119764804840085,
227
+ "step": 310
228
+ },
229
+ {
230
+ "epoch": 1.1351111111111112,
231
+ "grad_norm": 0.1756429374217987,
232
+ "learning_rate": 0.0001568587320909255,
233
+ "loss": 0.0650447130203247,
234
+ "step": 320
235
+ },
236
+ {
237
+ "epoch": 1.1706666666666667,
238
+ "grad_norm": 0.1355263888835907,
239
+ "learning_rate": 0.00015341526606309645,
240
+ "loss": 0.06544734239578247,
241
+ "step": 330
242
+ },
243
+ {
244
+ "epoch": 1.2062222222222223,
245
+ "grad_norm": 0.20916008949279785,
246
+ "learning_rate": 0.00014988078059750652,
247
+ "loss": 0.06727538704872131,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 1.2417777777777779,
252
+ "grad_norm": 0.19566652178764343,
253
+ "learning_rate": 0.00014626129844576893,
254
+ "loss": 0.06408223509788513,
255
+ "step": 350
256
+ },
257
+ {
258
+ "epoch": 1.2773333333333334,
259
+ "grad_norm": 0.05987081304192543,
260
+ "learning_rate": 0.00014256298719357062,
261
+ "loss": 0.04535002112388611,
262
+ "step": 360
263
+ },
264
+ {
265
+ "epoch": 1.3128888888888888,
266
+ "grad_norm": 0.1902439296245575,
267
+ "learning_rate": 0.00013879214875112665,
268
+ "loss": 0.04922315180301666,
269
+ "step": 370
270
+ },
271
+ {
272
+ "epoch": 1.3484444444444446,
273
+ "grad_norm": 0.25207415223121643,
274
+ "learning_rate": 0.00013495520861474565,
275
+ "loss": 0.060137057304382326,
276
+ "step": 380
277
+ },
278
+ {
279
+ "epoch": 1.384,
280
+ "grad_norm": 0.34174537658691406,
281
+ "learning_rate": 0.00013105870491780558,
282
+ "loss": 0.05463656783103943,
283
+ "step": 390
284
+ },
285
+ {
286
+ "epoch": 1.4195555555555557,
287
+ "grad_norm": 0.19624291360378265,
288
+ "learning_rate": 0.00012710927728979568,
289
+ "loss": 0.06046912670135498,
290
+ "step": 400
291
+ },
292
+ {
293
+ "epoch": 1.455111111111111,
294
+ "grad_norm": 0.22849531471729279,
295
+ "learning_rate": 0.00012311365554240971,
296
+ "loss": 0.039173880219459535,
297
+ "step": 410
298
+ },
299
+ {
300
+ "epoch": 1.4906666666666666,
301
+ "grad_norm": 0.22941145300865173,
302
+ "learning_rate": 0.0001190786482019691,
303
+ "loss": 0.04884783029556274,
304
+ "step": 420
305
+ },
306
+ {
307
+ "epoch": 1.5262222222222221,
308
+ "grad_norm": 0.2867370843887329,
309
+ "learning_rate": 0.00011501113090771619,
310
+ "loss": 0.05086652636528015,
311
+ "step": 430
312
+ },
313
+ {
314
+ "epoch": 1.561777777777778,
315
+ "grad_norm": 0.27643826603889465,
316
+ "learning_rate": 0.00011091803469574789,
317
+ "loss": 0.04540249407291412,
318
+ "step": 440
319
+ },
320
+ {
321
+ "epoch": 1.5973333333333333,
322
+ "grad_norm": 0.1365540772676468,
323
+ "learning_rate": 0.00010680633418855267,
324
+ "loss": 0.0644813358783722,
325
+ "step": 450
326
+ },
327
+ {
328
+ "epoch": 1.6328888888888888,
329
+ "grad_norm": 0.21614767611026764,
330
+ "learning_rate": 0.00010268303571027696,
331
+ "loss": 0.07455622553825378,
332
+ "step": 460
333
+ },
334
+ {
335
+ "epoch": 1.6684444444444444,
336
+ "grad_norm": 0.08498027920722961,
337
+ "learning_rate": 9.855516534797187e-05,
338
+ "loss": 0.04775569438934326,
339
+ "step": 470
340
+ },
341
+ {
342
+ "epoch": 1.704,
343
+ "grad_norm": 0.2029683142900467,
344
+ "learning_rate": 9.442975697916372e-05,
345
+ "loss": 0.04402676820755005,
346
+ "step": 480
347
+ },
348
+ {
349
+ "epoch": 1.7395555555555555,
350
+ "grad_norm": 0.34462177753448486,
351
+ "learning_rate": 9.031384028615004e-05,
352
+ "loss": 0.05336908102035522,
353
+ "step": 490
354
+ },
355
+ {
356
+ "epoch": 1.775111111111111,
357
+ "grad_norm": 0.05993572995066643,
358
+ "learning_rate": 8.621442877744409e-05,
359
+ "loss": 0.04580667018890381,
360
+ "step": 500
361
+ },
362
+ {
363
+ "epoch": 1.775111111111111,
364
+ "eval_accuracy": 0.9426666666666668,
365
+ "eval_loss": 0.15560266375541687,
366
+ "eval_mcq_accuracy": 0.7222222222222222,
367
+ "eval_runtime": 26.3716,
368
+ "eval_samples_per_second": 17.064,
369
+ "eval_steps_per_second": 4.285,
370
+ "step": 500
371
+ },
372
+ {
373
+ "epoch": 1.8106666666666666,
374
+ "grad_norm": 0.3891923427581787,
375
+ "learning_rate": 8.213850783677925e-05,
376
+ "loss": 0.03394646048545837,
377
+ "step": 510
378
+ },
379
+ {
380
+ "epoch": 1.8462222222222222,
381
+ "grad_norm": 0.1975124329328537,
382
+ "learning_rate": 7.809302282003823e-05,
383
+ "loss": 0.054677408933639524,
384
+ "step": 520
385
+ },
386
+ {
387
+ "epoch": 1.8817777777777778,
388
+ "grad_norm": 0.1962609589099884,
389
+ "learning_rate": 7.408486722038943e-05,
390
+ "loss": 0.06665679812431335,
391
+ "step": 530
392
+ },
393
+ {
394
+ "epoch": 1.9173333333333333,
395
+ "grad_norm": 0.2969353199005127,
396
+ "learning_rate": 7.012087092179724e-05,
397
+ "loss": 0.048915204405784604,
398
+ "step": 540
399
+ },
400
+ {
401
+ "epoch": 1.952888888888889,
402
+ "grad_norm": 0.23223088681697845,
403
+ "learning_rate": 6.620778856092227e-05,
404
+ "loss": 0.04671376347541809,
405
+ "step": 550
406
+ },
407
+ {
408
+ "epoch": 1.9884444444444445,
409
+ "grad_norm": 0.4542539417743683,
410
+ "learning_rate": 6.235228801724253e-05,
411
+ "loss": 0.05778223276138306,
412
+ "step": 560
413
+ },
414
+ {
415
+ "epoch": 2.021333333333333,
416
+ "grad_norm": 0.09194125235080719,
417
+ "learning_rate": 5.856093905100899e-05,
418
+ "loss": 0.022845838963985444,
419
+ "step": 570
420
+ },
421
+ {
422
+ "epoch": 2.056888888888889,
423
+ "grad_norm": 0.10670984536409378,
424
+ "learning_rate": 5.4840202108395466e-05,
425
+ "loss": 0.00663934126496315,
426
+ "step": 580
427
+ },
428
+ {
429
+ "epoch": 2.0924444444444443,
430
+ "grad_norm": 0.015582868829369545,
431
+ "learning_rate": 5.119641731291971e-05,
432
+ "loss": 0.007281148433685302,
433
+ "step": 590
434
+ },
435
+ {
436
+ "epoch": 2.128,
437
+ "grad_norm": 0.17645655572414398,
438
+ "learning_rate": 4.7635793661893666e-05,
439
+ "loss": 0.00572865828871727,
440
+ "step": 600
441
+ },
442
+ {
443
+ "epoch": 2.1635555555555555,
444
+ "grad_norm": 0.46430906653404236,
445
+ "learning_rate": 4.416439844631271e-05,
446
+ "loss": 0.008638855814933778,
447
+ "step": 610
448
+ },
449
+ {
450
+ "epoch": 2.1991111111111112,
451
+ "grad_norm": 0.011537984013557434,
452
+ "learning_rate": 4.078814691221139e-05,
453
+ "loss": 0.004736468568444252,
454
+ "step": 620
455
+ },
456
+ {
457
+ "epoch": 2.2346666666666666,
458
+ "grad_norm": 0.00896433461457491,
459
+ "learning_rate": 3.751279218110387e-05,
460
+ "loss": 0.007740923762321472,
461
+ "step": 630
462
+ },
463
+ {
464
+ "epoch": 2.2702222222222224,
465
+ "grad_norm": 0.002848990960046649,
466
+ "learning_rate": 3.434391544668383e-05,
467
+ "loss": 0.007571302354335785,
468
+ "step": 640
469
+ },
470
+ {
471
+ "epoch": 2.3057777777777777,
472
+ "grad_norm": 0.01570839062333107,
473
+ "learning_rate": 3.1286916464488505e-05,
474
+ "loss": 0.006724107265472412,
475
+ "step": 650
476
+ },
477
+ {
478
+ "epoch": 2.3413333333333335,
479
+ "grad_norm": 0.003965158946812153,
480
+ "learning_rate": 2.8347004350733185e-05,
481
+ "loss": 0.008114227652549743,
482
+ "step": 660
483
+ },
484
+ {
485
+ "epoch": 2.376888888888889,
486
+ "grad_norm": 0.10036994516849518,
487
+ "learning_rate": 2.55291887059944e-05,
488
+ "loss": 0.0027894463390111925,
489
+ "step": 670
490
+ },
491
+ {
492
+ "epoch": 2.4124444444444446,
493
+ "grad_norm": 0.1702636480331421,
494
+ "learning_rate": 2.2838271078866714e-05,
495
+ "loss": 0.0071901664137840274,
496
+ "step": 680
497
+ },
498
+ {
499
+ "epoch": 2.448,
500
+ "grad_norm": 0.003337263595312834,
501
+ "learning_rate": 2.0278836784140044e-05,
502
+ "loss": 0.0033892091363668443,
503
+ "step": 690
504
+ },
505
+ {
506
+ "epoch": 2.4835555555555557,
507
+ "grad_norm": 0.0054590729996562,
508
+ "learning_rate": 1.785524708943802e-05,
509
+ "loss": 0.002477823756635189,
510
+ "step": 700
511
+ },
512
+ {
513
+ "epoch": 2.519111111111111,
514
+ "grad_norm": 0.02814805507659912,
515
+ "learning_rate": 1.557163178363251e-05,
516
+ "loss": 0.0032980531454086305,
517
+ "step": 710
518
+ },
519
+ {
520
+ "epoch": 2.554666666666667,
521
+ "grad_norm": 0.01721755601465702,
522
+ "learning_rate": 1.3431882139696916e-05,
523
+ "loss": 0.00092921182513237,
524
+ "step": 720
525
+ },
526
+ {
527
+ "epoch": 2.590222222222222,
528
+ "grad_norm": 0.004700134973973036,
529
+ "learning_rate": 1.1439644283989747e-05,
530
+ "loss": 0.008682060241699218,
531
+ "step": 730
532
+ },
533
+ {
534
+ "epoch": 2.6257777777777775,
535
+ "grad_norm": 0.02318074181675911,
536
+ "learning_rate": 9.59831298326731e-06,
537
+ "loss": 0.0035014577209949494,
538
+ "step": 740
539
+ },
540
+ {
541
+ "epoch": 2.6613333333333333,
542
+ "grad_norm": 0.10873377323150635,
543
+ "learning_rate": 7.911025860012444e-06,
544
+ "loss": 0.007874777913093567,
545
+ "step": 750
546
+ },
547
+ {
548
+ "epoch": 2.696888888888889,
549
+ "grad_norm": 0.007307600695639849,
550
+ "learning_rate": 6.38065804593595e-06,
551
+ "loss": 0.003820793330669403,
552
+ "step": 760
553
+ },
554
+ {
555
+ "epoch": 2.7324444444444445,
556
+ "grad_norm": 0.002649878617376089,
557
+ "learning_rate": 5.009817282761675e-06,
558
+ "loss": 0.0014227939769625663,
559
+ "step": 770
560
+ },
561
+ {
562
+ "epoch": 2.768,
563
+ "grad_norm": 0.02097943052649498,
564
+ "learning_rate": 3.800839478643259e-06,
565
+ "loss": 0.007305952161550522,
566
+ "step": 780
567
+ },
568
+ {
569
+ "epoch": 2.8035555555555556,
570
+ "grad_norm": 0.11966679990291595,
571
+ "learning_rate": 2.7557847277841942e-06,
572
+ "loss": 0.004338302463293075,
573
+ "step": 790
574
+ },
575
+ {
576
+ "epoch": 2.8391111111111114,
577
+ "grad_norm": 0.07167425006628036,
578
+ "learning_rate": 1.8764338000442083e-06,
579
+ "loss": 0.002496876008808613,
580
+ "step": 800
581
+ },
582
+ {
583
+ "epoch": 2.8746666666666667,
584
+ "grad_norm": 0.010830031707882881,
585
+ "learning_rate": 1.1642851065131633e-06,
586
+ "loss": 0.0015433511696755886,
587
+ "step": 810
588
+ },
589
+ {
590
+ "epoch": 2.910222222222222,
591
+ "grad_norm": 0.195694237947464,
592
+ "learning_rate": 6.205521462235186e-07,
593
+ "loss": 0.001970846764743328,
594
+ "step": 820
595
+ },
596
+ {
597
+ "epoch": 2.945777777777778,
598
+ "grad_norm": 0.00489334249868989,
599
+ "learning_rate": 2.4616143835202166e-07,
600
+ "loss": 0.010426017642021179,
601
+ "step": 830
602
+ },
603
+ {
604
+ "epoch": 2.981333333333333,
605
+ "grad_norm": 0.009853039868175983,
606
+ "learning_rate": 4.1750943434026855e-08,
607
+ "loss": 0.0070929393172264096,
608
+ "step": 840
609
+ },
610
+ {
611
+ "epoch": 3.0,
612
+ "step": 846,
613
+ "total_flos": 1.0958550940031386e+17,
614
+ "train_loss": 0.06952556652285546,
615
+ "train_runtime": 2162.4382,
616
+ "train_samples_per_second": 6.243,
617
+ "train_steps_per_second": 0.391
618
+ }
619
+ ],
620
+ "logging_steps": 10,
621
+ "max_steps": 846,
622
+ "num_input_tokens_seen": 0,
623
+ "num_train_epochs": 3,
624
+ "save_steps": 500,
625
+ "stateful_callbacks": {
626
+ "TrainerControl": {
627
+ "args": {
628
+ "should_epoch_stop": false,
629
+ "should_evaluate": false,
630
+ "should_log": false,
631
+ "should_save": true,
632
+ "should_training_stop": true
633
+ },
634
+ "attributes": {}
635
+ }
636
+ },
637
+ "total_flos": 1.0958550940031386e+17,
638
+ "train_batch_size": 4,
639
+ "trial_name": null,
640
+ "trial_params": null
641
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06e72e4c171697e89ed577da3f438cb2d6210bd6fe361313102cda2f6b7351b8
3
+ size 5649
training_eval_accuracy.png ADDED
training_eval_loss.png ADDED
training_loss.png ADDED