Raiff1982 commited on
Commit
7f4cf8e
·
verified ·
1 Parent(s): 3e2545e

Upload folder using huggingface_hub

Browse files
newton/README.md CHANGED
@@ -1,9 +1,9 @@
1
  ---
2
- base_model: meta-llama/Llama-3.1-8B-Instruct
3
  library_name: peft
4
  model_name: newton
5
  tags:
6
- - base_model:adapter:meta-llama/Llama-3.1-8B-Instruct
7
  - lora
8
  - sft
9
  - transformers
@@ -14,7 +14,7 @@ pipeline_tag: text-generation
14
 
15
  # Model Card for newton
16
 
17
- This model is a fine-tuned version of [meta-llama/Llama-3.1-8B-Instruct](https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct).
18
  It has been trained using [TRL](https://github.com/huggingface/trl).
19
 
20
  ## Quick start
@@ -39,10 +39,10 @@ This model was trained with SFT.
39
  ### Framework versions
40
 
41
  - PEFT 0.18.1
42
- - TRL: 0.29.1
43
- - Transformers: 5.3.0
44
- - Pytorch: 2.10.0
45
- - Datasets: 4.8.3
46
  - Tokenizers: 0.22.2
47
 
48
  ## Citations
 
1
  ---
2
+ base_model: Raiff1982/codette-llama-3.1-8b-merged
3
  library_name: peft
4
  model_name: newton
5
  tags:
6
+ - base_model:adapter:Raiff1982/codette-llama-3.1-8b-merged
7
  - lora
8
  - sft
9
  - transformers
 
14
 
15
  # Model Card for newton
16
 
17
+ This model is a fine-tuned version of [Raiff1982/codette-llama-3.1-8b-merged](https://huggingface.co/Raiff1982/codette-llama-3.1-8b-merged).
18
  It has been trained using [TRL](https://github.com/huggingface/trl).
19
 
20
  ## Quick start
 
39
  ### Framework versions
40
 
41
  - PEFT 0.18.1
42
+ - TRL: 1.0.0
43
+ - Transformers: 5.5.0
44
+ - Pytorch: 2.11.0
45
+ - Datasets: 4.8.4
46
  - Tokenizers: 0.22.2
47
 
48
  ## Citations
newton/adapter_config.json CHANGED
@@ -3,7 +3,7 @@
3
  "alpha_pattern": {},
4
  "arrow_config": null,
5
  "auto_mapping": null,
6
- "base_model_name_or_path": "meta-llama/Llama-3.1-8B-Instruct",
7
  "bias": "none",
8
  "corda_config": null,
9
  "ensure_weight_tying": false,
@@ -29,10 +29,10 @@
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
32
- "o_proj",
33
- "v_proj",
34
  "q_proj",
35
- "k_proj"
 
 
36
  ],
37
  "target_parameters": null,
38
  "task_type": "CAUSAL_LM",
 
3
  "alpha_pattern": {},
4
  "arrow_config": null,
5
  "auto_mapping": null,
6
+ "base_model_name_or_path": "Raiff1982/codette-llama-3.1-8b-merged",
7
  "bias": "none",
8
  "corda_config": null,
9
  "ensure_weight_tying": false,
 
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
 
 
32
  "q_proj",
33
+ "k_proj",
34
+ "o_proj",
35
+ "v_proj"
36
  ],
37
  "target_parameters": null,
38
  "task_type": "CAUSAL_LM",
newton/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15e7ab5a4223f583a958b59b89be7c7a7a2892c32c63e7d8880e92c28582f30e
3
  size 27297544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e66a75cae8fde6774170b0e340f267894de55426775f0c1693a12ce630003af8
3
  size 27297544
newton/checkpoint-309/README.md CHANGED
@@ -1,9 +1,9 @@
1
  ---
2
- base_model: meta-llama/Llama-3.1-8B-Instruct
3
  library_name: peft
4
  pipeline_tag: text-generation
5
  tags:
6
- - base_model:adapter:meta-llama/Llama-3.1-8B-Instruct
7
  - lora
8
  - sft
9
  - transformers
 
1
  ---
2
+ base_model: Raiff1982/codette-llama-3.1-8b-merged
3
  library_name: peft
4
  pipeline_tag: text-generation
5
  tags:
6
+ - base_model:adapter:Raiff1982/codette-llama-3.1-8b-merged
7
  - lora
8
  - sft
9
  - transformers
newton/checkpoint-309/adapter_config.json CHANGED
@@ -3,7 +3,7 @@
3
  "alpha_pattern": {},
4
  "arrow_config": null,
5
  "auto_mapping": null,
6
- "base_model_name_or_path": "meta-llama/Llama-3.1-8B-Instruct",
7
  "bias": "none",
8
  "corda_config": null,
9
  "ensure_weight_tying": false,
@@ -29,10 +29,10 @@
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
32
- "o_proj",
33
- "v_proj",
34
  "q_proj",
35
- "k_proj"
 
 
36
  ],
37
  "target_parameters": null,
38
  "task_type": "CAUSAL_LM",
 
3
  "alpha_pattern": {},
4
  "arrow_config": null,
5
  "auto_mapping": null,
6
+ "base_model_name_or_path": "Raiff1982/codette-llama-3.1-8b-merged",
7
  "bias": "none",
8
  "corda_config": null,
9
  "ensure_weight_tying": false,
 
29
  "rank_pattern": {},
30
  "revision": null,
31
  "target_modules": [
 
 
32
  "q_proj",
33
+ "k_proj",
34
+ "o_proj",
35
+ "v_proj"
36
  ],
37
  "target_parameters": null,
38
  "task_type": "CAUSAL_LM",
newton/checkpoint-309/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:15e7ab5a4223f583a958b59b89be7c7a7a2892c32c63e7d8880e92c28582f30e
3
  size 27297544
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e66a75cae8fde6774170b0e340f267894de55426775f0c1693a12ce630003af8
3
  size 27297544
newton/checkpoint-309/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c0048ede16e73e75ae0558a8be34314fdf4dd7a0048a2c0345323e009f8cd75a
3
  size 54745547
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a9cc12bcba29c33eaebecec0edb4b744ab01f0d521051ab6ce9ac51ff823612a
3
  size 54745547
newton/checkpoint-309/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7a3ce3a6237917355c7ff6bb395ee6541fd028f6e898cd29ab73f2c1037e36ca
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:167c516cb3da9c574bcafa89ab7815d70767a2593edafa4119230d61545ccb9f
3
  size 14645
newton/checkpoint-309/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:dfae2b0e67a84fe0276ee2a4bd443baa818fcadf7c94eb154f2c9f40764c3223
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b805c703bca5d31d07f2c000dd76db9e70297bbf6528c7f0fe3645a9d4d5a44c
3
  size 1465
newton/checkpoint-309/trainer_state.json CHANGED
@@ -10,303 +10,303 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "entropy": 2.5801601350307464,
14
- "epoch": 0.0975609756097561,
15
- "grad_norm": 0.640625,
16
- "learning_rate": 0.00018,
17
- "loss": 3.2993820190429686,
18
- "mean_token_accuracy": 0.4129736639559269,
19
- "num_tokens": 20093.0,
20
  "step": 10
21
  },
22
  {
23
- "entropy": 1.46146529763937,
24
- "epoch": 0.1951219512195122,
25
- "grad_norm": 0.7109375,
26
- "learning_rate": 0.0001939799331103679,
27
- "loss": 1.3410646438598632,
28
- "mean_token_accuracy": 0.6868809185922146,
29
- "num_tokens": 40396.0,
30
  "step": 20
31
  },
32
  {
33
- "entropy": 0.30421045832335947,
34
- "epoch": 0.2926829268292683,
35
- "grad_norm": 0.1591796875,
36
- "learning_rate": 0.00018729096989966558,
37
- "loss": 0.2733942985534668,
38
- "mean_token_accuracy": 0.9430402711033821,
39
- "num_tokens": 60559.0,
40
  "step": 30
41
  },
42
  {
43
- "entropy": 0.1860255692154169,
44
- "epoch": 0.3902439024390244,
45
- "grad_norm": 0.11865234375,
46
- "learning_rate": 0.00018060200668896322,
47
- "loss": 0.17133373022079468,
48
- "mean_token_accuracy": 0.9654295533895493,
49
- "num_tokens": 80726.0,
50
  "step": 40
51
  },
52
  {
53
- "entropy": 0.16763325594365597,
54
- "epoch": 0.4878048780487805,
55
- "grad_norm": 0.11328125,
56
- "learning_rate": 0.00017391304347826088,
57
- "loss": 0.1414021611213684,
58
- "mean_token_accuracy": 0.9735398545861245,
59
- "num_tokens": 100934.0,
60
  "step": 50
61
  },
62
  {
63
- "entropy": 0.14396543558686972,
64
- "epoch": 0.5853658536585366,
65
- "grad_norm": 0.1025390625,
66
- "learning_rate": 0.00016722408026755855,
67
- "loss": 0.11398149728775024,
68
- "mean_token_accuracy": 0.9769163399934768,
69
- "num_tokens": 121033.0,
70
  "step": 60
71
  },
72
  {
73
- "entropy": 0.09721304103732109,
74
- "epoch": 0.6829268292682927,
75
- "grad_norm": 0.1484375,
76
- "learning_rate": 0.00016053511705685619,
77
- "loss": 0.09096546769142151,
78
- "mean_token_accuracy": 0.9776118725538254,
79
- "num_tokens": 141112.0,
80
  "step": 70
81
  },
82
  {
83
- "entropy": 0.10199905186891556,
84
- "epoch": 0.7804878048780488,
85
- "grad_norm": 0.09765625,
86
- "learning_rate": 0.00015384615384615385,
87
- "loss": 0.07685146927833557,
88
- "mean_token_accuracy": 0.9789551600813866,
89
- "num_tokens": 161349.0,
90
  "step": 80
91
  },
92
  {
93
- "entropy": 0.09305545147508383,
94
- "epoch": 0.8780487804878049,
95
- "grad_norm": 0.0869140625,
96
- "learning_rate": 0.0001471571906354515,
97
- "loss": 0.06557443141937255,
98
- "mean_token_accuracy": 0.9781694248318672,
99
- "num_tokens": 181560.0,
100
  "step": 90
101
  },
102
  {
103
- "entropy": 0.0788642093539238,
104
- "epoch": 0.975609756097561,
105
- "grad_norm": 0.051025390625,
106
- "learning_rate": 0.00014046822742474916,
107
- "loss": 0.05446377396583557,
108
- "mean_token_accuracy": 0.9795627012848854,
109
- "num_tokens": 201709.0,
110
  "step": 100
111
  },
112
  {
113
- "entropy": 0.06586153550367606,
114
- "epoch": 1.0682926829268293,
115
- "grad_norm": 0.056640625,
116
- "learning_rate": 0.00013377926421404682,
117
- "loss": 0.05286722779273987,
118
- "mean_token_accuracy": 0.9784782999440244,
119
- "num_tokens": 220911.0,
120
  "step": 110
121
  },
122
  {
123
- "entropy": 0.06249262308701873,
124
- "epoch": 1.1658536585365853,
125
- "grad_norm": 0.0498046875,
126
- "learning_rate": 0.0001270903010033445,
127
- "loss": 0.052178531885147095,
128
- "mean_token_accuracy": 0.9783070877194404,
129
- "num_tokens": 241100.0,
130
  "step": 120
131
  },
132
  {
133
- "entropy": 0.05583516648039222,
134
- "epoch": 1.2634146341463415,
135
- "grad_norm": 0.0576171875,
136
- "learning_rate": 0.00012040133779264215,
137
- "loss": 0.04987884163856506,
138
- "mean_token_accuracy": 0.9790923863649368,
139
- "num_tokens": 261282.0,
140
  "step": 130
141
  },
142
  {
143
- "entropy": 0.05573393665254116,
144
- "epoch": 1.3609756097560974,
145
- "grad_norm": 0.05078125,
146
- "learning_rate": 0.00011371237458193979,
147
- "loss": 0.049983879923820494,
148
- "mean_token_accuracy": 0.9795695841312408,
149
- "num_tokens": 281470.0,
150
  "step": 140
151
  },
152
  {
153
- "entropy": 0.05604505110532045,
154
- "epoch": 1.4585365853658536,
155
- "grad_norm": 0.049560546875,
156
- "learning_rate": 0.00010702341137123746,
157
- "loss": 0.049191102385520935,
158
- "mean_token_accuracy": 0.9789946660399437,
159
- "num_tokens": 301537.0,
160
  "step": 150
161
  },
162
  {
163
- "entropy": 0.05383661286905408,
164
- "epoch": 1.5560975609756098,
165
- "grad_norm": 0.052001953125,
166
- "learning_rate": 0.00010033444816053512,
167
- "loss": 0.04975903928279877,
168
- "mean_token_accuracy": 0.9779537573456765,
169
- "num_tokens": 321687.0,
170
  "step": 160
171
  },
172
  {
173
- "entropy": 0.05550122009590268,
174
- "epoch": 1.653658536585366,
175
- "grad_norm": 0.05517578125,
176
- "learning_rate": 9.364548494983279e-05,
177
- "loss": 0.04978601038455963,
178
- "mean_token_accuracy": 0.9779243767261505,
179
- "num_tokens": 341947.0,
180
  "step": 170
181
  },
182
  {
183
- "entropy": 0.0521472885273397,
184
- "epoch": 1.751219512195122,
185
- "grad_norm": 0.053466796875,
186
- "learning_rate": 8.695652173913044e-05,
187
- "loss": 0.04891734123229981,
188
- "mean_token_accuracy": 0.9793664410710334,
189
- "num_tokens": 362157.0,
190
  "step": 180
191
  },
192
  {
193
- "entropy": 0.05427801813930273,
194
- "epoch": 1.848780487804878,
195
- "grad_norm": 0.03564453125,
196
- "learning_rate": 8.026755852842809e-05,
197
- "loss": 0.04973878860473633,
198
- "mean_token_accuracy": 0.9789470374584198,
199
- "num_tokens": 382284.0,
200
  "step": 190
201
  },
202
  {
203
- "entropy": 0.05222779270261526,
204
- "epoch": 1.946341463414634,
205
- "grad_norm": 0.056884765625,
206
- "learning_rate": 7.357859531772575e-05,
207
- "loss": 0.04821131825447082,
208
- "mean_token_accuracy": 0.9792478010058403,
209
- "num_tokens": 402458.0,
210
  "step": 200
211
  },
212
  {
213
- "entropy": 0.0516971405595541,
214
- "epoch": 2.0390243902439025,
215
- "grad_norm": 0.04345703125,
216
- "learning_rate": 6.688963210702341e-05,
217
- "loss": 0.04751978814601898,
218
- "mean_token_accuracy": 0.9792861828678533,
219
- "num_tokens": 421706.0,
220
  "step": 210
221
  },
222
  {
223
- "entropy": 0.05148327555507422,
224
- "epoch": 2.1365853658536587,
225
- "grad_norm": 0.043701171875,
226
- "learning_rate": 6.0200668896321076e-05,
227
- "loss": 0.04758384227752686,
228
- "mean_token_accuracy": 0.9795778721570969,
229
- "num_tokens": 442021.0,
230
  "step": 220
231
  },
232
  {
233
- "entropy": 0.05236647073179483,
234
- "epoch": 2.234146341463415,
235
- "grad_norm": 0.040283203125,
236
- "learning_rate": 5.351170568561873e-05,
237
- "loss": 0.04777625799179077,
238
- "mean_token_accuracy": 0.9786792114377022,
239
- "num_tokens": 462138.0,
240
  "step": 230
241
  },
242
  {
243
- "entropy": 0.05254681948572397,
244
- "epoch": 2.3317073170731706,
245
- "grad_norm": 0.03857421875,
246
- "learning_rate": 4.6822742474916394e-05,
247
- "loss": 0.047863802313804625,
248
- "mean_token_accuracy": 0.9788415163755417,
249
- "num_tokens": 482318.0,
250
  "step": 240
251
  },
252
  {
253
- "entropy": 0.0514264177531004,
254
- "epoch": 2.4292682926829268,
255
- "grad_norm": 0.041259765625,
256
- "learning_rate": 4.0133779264214046e-05,
257
- "loss": 0.04747449159622193,
258
- "mean_token_accuracy": 0.9798328444361687,
259
- "num_tokens": 502473.0,
260
  "step": 250
261
  },
262
  {
263
- "entropy": 0.05115404771640897,
264
- "epoch": 2.526829268292683,
265
- "grad_norm": 0.0576171875,
266
- "learning_rate": 3.3444816053511705e-05,
267
- "loss": 0.04702305793762207,
268
- "mean_token_accuracy": 0.9793446376919747,
269
- "num_tokens": 522684.0,
270
  "step": 260
271
  },
272
  {
273
- "entropy": 0.05094934357330203,
274
- "epoch": 2.624390243902439,
275
- "grad_norm": 0.052734375,
276
- "learning_rate": 2.6755852842809364e-05,
277
- "loss": 0.04771294593811035,
278
- "mean_token_accuracy": 0.9783476650714874,
279
- "num_tokens": 542810.0,
280
  "step": 270
281
  },
282
  {
283
- "entropy": 0.0513969948515296,
284
- "epoch": 2.721951219512195,
285
- "grad_norm": 0.0517578125,
286
- "learning_rate": 2.0066889632107023e-05,
287
- "loss": 0.04680090546607971,
288
- "mean_token_accuracy": 0.9800771608948707,
289
- "num_tokens": 563051.0,
290
  "step": 280
291
  },
292
  {
293
- "entropy": 0.05136826252564788,
294
- "epoch": 2.819512195121951,
295
- "grad_norm": 0.0400390625,
296
- "learning_rate": 1.3377926421404682e-05,
297
- "loss": 0.0476355105638504,
298
- "mean_token_accuracy": 0.9787444621324539,
299
- "num_tokens": 583158.0,
300
  "step": 290
301
  },
302
  {
303
- "entropy": 0.05182668277993798,
304
- "epoch": 2.9170731707317072,
305
- "grad_norm": 0.045654296875,
306
- "learning_rate": 6.688963210702341e-06,
307
- "loss": 0.04734614789485932,
308
- "mean_token_accuracy": 0.978977257013321,
309
- "num_tokens": 603264.0,
310
  "step": 300
311
  }
312
  ],
@@ -327,7 +327,7 @@
327
  "attributes": {}
328
  }
329
  },
330
- "total_flos": 2.857929508159488e+16,
331
  "train_batch_size": 2,
332
  "trial_name": null,
333
  "trial_params": null
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "entropy": 2.487887626886368,
14
+ "epoch": 0.0970873786407767,
15
+ "grad_norm": 0.55078125,
16
+ "learning_rate": 9e-05,
17
+ "loss": 3.4457256317138674,
18
+ "mean_token_accuracy": 0.4037958011031151,
19
+ "num_tokens": 20045.0,
20
  "step": 10
21
  },
22
  {
23
+ "entropy": 2.367420971393585,
24
+ "epoch": 0.1941747572815534,
25
+ "grad_norm": 1.03125,
26
+ "learning_rate": 9.698996655518396e-05,
27
+ "loss": 2.301749038696289,
28
+ "mean_token_accuracy": 0.5057214729487896,
29
+ "num_tokens": 40110.0,
30
  "step": 20
31
  },
32
  {
33
+ "entropy": 0.9785687401890755,
34
+ "epoch": 0.2912621359223301,
35
+ "grad_norm": 0.482421875,
36
+ "learning_rate": 9.364548494983279e-05,
37
+ "loss": 0.8281048774719239,
38
+ "mean_token_accuracy": 0.8013610184192658,
39
+ "num_tokens": 60360.0,
40
  "step": 30
41
  },
42
  {
43
+ "entropy": 0.3154874790459871,
44
+ "epoch": 0.3883495145631068,
45
+ "grad_norm": 0.275390625,
46
+ "learning_rate": 9.030100334448161e-05,
47
+ "loss": 0.3118274450302124,
48
+ "mean_token_accuracy": 0.936956076323986,
49
+ "num_tokens": 80424.0,
50
  "step": 40
51
  },
52
  {
53
+ "entropy": 0.21394508741796017,
54
+ "epoch": 0.4854368932038835,
55
+ "grad_norm": 0.2236328125,
56
+ "learning_rate": 8.695652173913044e-05,
57
+ "loss": 0.1942020535469055,
58
+ "mean_token_accuracy": 0.9629994183778763,
59
+ "num_tokens": 100413.0,
60
  "step": 50
61
  },
62
  {
63
+ "entropy": 0.17603826858103275,
64
+ "epoch": 0.5825242718446602,
65
+ "grad_norm": 0.162109375,
66
+ "learning_rate": 8.361204013377927e-05,
67
+ "loss": 0.17068777084350586,
68
+ "mean_token_accuracy": 0.9660190954804421,
69
+ "num_tokens": 120668.0,
70
  "step": 60
71
  },
72
  {
73
+ "entropy": 0.1720115825533867,
74
+ "epoch": 0.6796116504854369,
75
+ "grad_norm": 0.1533203125,
76
+ "learning_rate": 8.026755852842809e-05,
77
+ "loss": 0.15503238439559935,
78
+ "mean_token_accuracy": 0.9691305428743362,
79
+ "num_tokens": 140907.0,
80
  "step": 70
81
  },
82
  {
83
+ "entropy": 0.15427112840116025,
84
+ "epoch": 0.7766990291262136,
85
+ "grad_norm": 0.177734375,
86
+ "learning_rate": 7.692307692307693e-05,
87
+ "loss": 0.14039652347564696,
88
+ "mean_token_accuracy": 0.9698389366269111,
89
+ "num_tokens": 161024.0,
90
  "step": 80
91
  },
92
  {
93
+ "entropy": 0.15376258343458177,
94
+ "epoch": 0.8737864077669902,
95
+ "grad_norm": 0.130859375,
96
+ "learning_rate": 7.357859531772575e-05,
97
+ "loss": 0.1298056960105896,
98
+ "mean_token_accuracy": 0.9742556199431419,
99
+ "num_tokens": 181243.0,
100
  "step": 90
101
  },
102
  {
103
+ "entropy": 0.14775074161589147,
104
+ "epoch": 0.970873786407767,
105
+ "grad_norm": 0.126953125,
106
+ "learning_rate": 7.023411371237458e-05,
107
+ "loss": 0.12051811218261718,
108
+ "mean_token_accuracy": 0.9776876136660576,
109
+ "num_tokens": 201412.0,
110
  "step": 100
111
  },
112
  {
113
+ "entropy": 0.15358976498246193,
114
+ "epoch": 1.0679611650485437,
115
+ "grad_norm": 0.1357421875,
116
+ "learning_rate": 6.688963210702341e-05,
117
+ "loss": 0.13191524744033814,
118
+ "mean_token_accuracy": 0.9737549498677254,
119
+ "num_tokens": 221317.0,
120
  "step": 110
121
  },
122
  {
123
+ "entropy": 0.12130333054810763,
124
+ "epoch": 1.1650485436893203,
125
+ "grad_norm": 0.111328125,
126
+ "learning_rate": 6.354515050167224e-05,
127
+ "loss": 0.09631805419921875,
128
+ "mean_token_accuracy": 0.9778564915060997,
129
+ "num_tokens": 241469.0,
130
  "step": 120
131
  },
132
  {
133
+ "entropy": 0.12368765268474817,
134
+ "epoch": 1.262135922330097,
135
+ "grad_norm": 0.126953125,
136
+ "learning_rate": 6.0200668896321076e-05,
137
+ "loss": 0.10474776029586792,
138
+ "mean_token_accuracy": 0.9744847908616066,
139
+ "num_tokens": 261431.0,
140
  "step": 130
141
  },
142
  {
143
+ "entropy": 0.10261523667722941,
144
+ "epoch": 1.3592233009708738,
145
+ "grad_norm": 0.11572265625,
146
+ "learning_rate": 5.6856187290969896e-05,
147
+ "loss": 0.076785808801651,
148
+ "mean_token_accuracy": 0.9786087408661842,
149
+ "num_tokens": 281589.0,
150
  "step": 140
151
  },
152
  {
153
+ "entropy": 0.09858085233718157,
154
+ "epoch": 1.4563106796116505,
155
+ "grad_norm": 0.10693359375,
156
+ "learning_rate": 5.351170568561873e-05,
157
+ "loss": 0.07057241201400757,
158
+ "mean_token_accuracy": 0.978242176771164,
159
+ "num_tokens": 301814.0,
160
  "step": 150
161
  },
162
  {
163
+ "entropy": 0.09467246811836957,
164
+ "epoch": 1.5533980582524272,
165
+ "grad_norm": 0.1171875,
166
+ "learning_rate": 5.016722408026756e-05,
167
+ "loss": 0.06553536057472228,
168
+ "mean_token_accuracy": 0.9788320794701576,
169
+ "num_tokens": 322034.0,
170
  "step": 160
171
  },
172
  {
173
+ "entropy": 0.09434448610991239,
174
+ "epoch": 1.650485436893204,
175
+ "grad_norm": 0.07666015625,
176
+ "learning_rate": 4.6822742474916394e-05,
177
+ "loss": 0.06265009045600892,
178
+ "mean_token_accuracy": 0.978448674082756,
179
+ "num_tokens": 342246.0,
180
  "step": 170
181
  },
182
  {
183
+ "entropy": 0.090944261290133,
184
+ "epoch": 1.7475728155339807,
185
+ "grad_norm": 0.10888671875,
186
+ "learning_rate": 4.347826086956522e-05,
187
+ "loss": 0.05846574306488037,
188
+ "mean_token_accuracy": 0.9787591323256493,
189
+ "num_tokens": 362493.0,
190
  "step": 180
191
  },
192
  {
193
+ "entropy": 0.10327567681670188,
194
+ "epoch": 1.8446601941747574,
195
+ "grad_norm": 0.09619140625,
196
+ "learning_rate": 4.0133779264214046e-05,
197
+ "loss": 0.07640682458877564,
198
+ "mean_token_accuracy": 0.9750214621424675,
199
+ "num_tokens": 382470.0,
200
  "step": 190
201
  },
202
  {
203
+ "entropy": 0.08134665545076132,
204
+ "epoch": 1.941747572815534,
205
+ "grad_norm": 0.08447265625,
206
+ "learning_rate": 3.678929765886287e-05,
207
+ "loss": 0.053827738761901854,
208
+ "mean_token_accuracy": 0.9790568739175797,
209
+ "num_tokens": 402761.0,
210
  "step": 200
211
  },
212
  {
213
+ "entropy": 0.09850292447954416,
214
+ "epoch": 2.0388349514563107,
215
+ "grad_norm": 0.1845703125,
216
+ "learning_rate": 3.3444816053511705e-05,
217
+ "loss": 0.0715977430343628,
218
+ "mean_token_accuracy": 0.9744585126638412,
219
+ "num_tokens": 422580.0,
220
  "step": 210
221
  },
222
  {
223
+ "entropy": 0.07439298946410418,
224
+ "epoch": 2.1359223300970873,
225
+ "grad_norm": 0.07763671875,
226
+ "learning_rate": 3.0100334448160538e-05,
227
+ "loss": 0.05062834024429321,
228
+ "mean_token_accuracy": 0.9796976149082184,
229
+ "num_tokens": 442815.0,
230
  "step": 220
231
  },
232
  {
233
+ "entropy": 0.0704705884680152,
234
+ "epoch": 2.233009708737864,
235
+ "grad_norm": 0.09716796875,
236
+ "learning_rate": 2.6755852842809364e-05,
237
+ "loss": 0.0514835000038147,
238
+ "mean_token_accuracy": 0.9794572740793228,
239
+ "num_tokens": 462923.0,
240
  "step": 230
241
  },
242
  {
243
+ "entropy": 0.07191200498491526,
244
+ "epoch": 2.3300970873786406,
245
+ "grad_norm": 0.07958984375,
246
+ "learning_rate": 2.3411371237458197e-05,
247
+ "loss": 0.051685428619384764,
248
+ "mean_token_accuracy": 0.9794511407613754,
249
+ "num_tokens": 483078.0,
250
  "step": 240
251
  },
252
  {
253
+ "entropy": 0.06878451202064753,
254
+ "epoch": 2.4271844660194173,
255
+ "grad_norm": 0.07470703125,
256
+ "learning_rate": 2.0066889632107023e-05,
257
+ "loss": 0.050105297565460206,
258
+ "mean_token_accuracy": 0.9797055780887604,
259
+ "num_tokens": 503252.0,
260
  "step": 250
261
  },
262
  {
263
+ "entropy": 0.08662745058536529,
264
+ "epoch": 2.524271844660194,
265
+ "grad_norm": 0.08837890625,
266
+ "learning_rate": 1.6722408026755853e-05,
267
+ "loss": 0.06598815321922302,
268
+ "mean_token_accuracy": 0.9757005825638772,
269
+ "num_tokens": 523400.0,
270
  "step": 260
271
  },
272
  {
273
+ "entropy": 0.08474288573488593,
274
+ "epoch": 2.6213592233009706,
275
+ "grad_norm": 0.0791015625,
276
+ "learning_rate": 1.3377926421404682e-05,
277
+ "loss": 0.0655213713645935,
278
+ "mean_token_accuracy": 0.976191833615303,
279
+ "num_tokens": 543416.0,
280
  "step": 270
281
  },
282
  {
283
+ "entropy": 0.06866553500294685,
284
+ "epoch": 2.7184466019417477,
285
+ "grad_norm": 0.072265625,
286
+ "learning_rate": 1.0033444816053512e-05,
287
+ "loss": 0.04991424381732941,
288
+ "mean_token_accuracy": 0.9802871778607368,
289
+ "num_tokens": 563552.0,
290
  "step": 280
291
  },
292
  {
293
+ "entropy": 0.06758161811158062,
294
+ "epoch": 2.8155339805825244,
295
+ "grad_norm": 0.07373046875,
296
+ "learning_rate": 6.688963210702341e-06,
297
+ "loss": 0.049510461091995236,
298
+ "mean_token_accuracy": 0.9791449457406998,
299
+ "num_tokens": 583870.0,
300
  "step": 290
301
  },
302
  {
303
+ "entropy": 0.06784124514088034,
304
+ "epoch": 2.912621359223301,
305
+ "grad_norm": 0.12890625,
306
+ "learning_rate": 3.3444816053511705e-06,
307
+ "loss": 0.04954342842102051,
308
+ "mean_token_accuracy": 0.9807449117302894,
309
+ "num_tokens": 603997.0,
310
  "step": 300
311
  }
312
  ],
 
327
  "attributes": {}
328
  }
329
  },
330
+ "total_flos": 2.867826935488512e+16,
331
  "train_batch_size": 2,
332
  "trial_name": null,
333
  "trial_params": null
newton/checkpoint-309/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8755273dccefb3d7fa41448d64a8c28d76451700a997d4cbd5f7ac202a091f77
3
- size 5585
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:acff9661de685950df198faf07a4951237f5448c33d774ab0fc779a61a3dd677
3
+ size 5649