CodeIsAbstract commited on
Commit
561e400
·
verified ·
1 Parent(s): 2ba7f04

Training in progress, step 200, checkpoint

Browse files
last-checkpoint/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4f3ff6417f337444900c9d0ee3418cae9650f976bbe8cad9be7e0ebb946b00b8
3
- size 329100200
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9ce65bbed9331f4da673d60d1dc63ba5675640ffcc17c86a99dd8472de8ef429
3
+ size 253577648
last-checkpoint/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:c37d5c1333fa20846532da34840740f4df14612f69d38d91951e2a09c63b6c97
3
- size 188845393
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:15cccd4cc3adad21271edce5de20158da491b15c14ccceddfd2093b1befc0139
3
+ size 37798173
last-checkpoint/rng_state.pth CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f5656667be1478f776aa5cbc3870cc2b7f43ab2e34c2fe808cae05f89d24863b
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8a9d93ac0e7384563f970848a6d911064afa1c50720cde8b4ec828d00b5723a5
3
  size 14645
last-checkpoint/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:17980325c36d869e063d721782688706e0e30902ea32568d55af1c62cc6c3e6f
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:002599c9fd9c2e34b48da3b08b765f3b266ceba26e25343d6c7012afd5618305
3
  size 1465
last-checkpoint/trainer_state.json CHANGED
@@ -2,7 +2,7 @@
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
- "epoch": 0.06666666666666667,
6
  "eval_steps": 50,
7
  "global_step": 200,
8
  "is_hyper_param_search": false,
@@ -10,320 +10,320 @@
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
- "epoch": 0.0016666666666666668,
14
- "grad_norm": 54.850189208984375,
15
- "learning_rate": 0.0002,
16
- "loss": 6.845032501220703,
17
  "step": 5
18
  },
19
  {
20
- "epoch": 0.0033333333333333335,
21
- "grad_norm": 55.685916900634766,
22
- "learning_rate": 0.00045000000000000004,
23
- "loss": 6.620301055908203,
24
  "step": 10
25
  },
26
  {
27
- "epoch": 0.005,
28
- "grad_norm": 7.728530406951904,
29
- "learning_rate": 0.0005,
30
- "loss": 7.703640747070312,
31
  "step": 15
32
  },
33
  {
34
- "epoch": 0.006666666666666667,
35
- "grad_norm": 7.862086296081543,
36
- "learning_rate": 0.0005,
37
- "loss": 7.1933952331542965,
38
  "step": 20
39
  },
40
  {
41
- "epoch": 0.008333333333333333,
42
- "grad_norm": 41934556.0,
43
- "learning_rate": 0.0005,
44
- "loss": 7.290315246582031,
45
  "step": 25
46
  },
47
  {
48
- "epoch": 0.01,
49
- "grad_norm": 32783185920.0,
50
- "learning_rate": 0.0005,
51
- "loss": 10.341773986816406,
52
  "step": 30
53
  },
54
  {
55
- "epoch": 0.011666666666666667,
56
- "grad_norm": 67137892352.0,
57
- "learning_rate": 0.0005,
58
- "loss": 16.722955322265626,
59
  "step": 35
60
  },
61
  {
62
- "epoch": 0.013333333333333334,
63
- "grad_norm": 107766048.0,
64
- "learning_rate": 0.0005,
65
- "loss": 18.834597778320312,
66
  "step": 40
67
  },
68
  {
69
- "epoch": 0.015,
70
- "grad_norm": 750511.1875,
71
- "learning_rate": 0.0005,
72
- "loss": 16.470211791992188,
73
  "step": 45
74
  },
75
  {
76
- "epoch": 0.016666666666666666,
77
- "grad_norm": 1690175.0,
78
- "learning_rate": 0.0005,
79
- "loss": 17.498741149902344,
80
  "step": 50
81
  },
82
  {
83
- "epoch": 0.016666666666666666,
84
- "eval_loss": 9.509819984436035,
85
- "eval_runtime": 217.3232,
86
- "eval_samples_per_second": 1.371,
87
- "eval_steps_per_second": 0.686,
88
  "step": 50
89
  },
90
  {
91
- "epoch": 0.018333333333333333,
92
- "grad_norm": 44012.49609375,
93
- "learning_rate": 0.0005,
94
- "loss": 19.190980529785158,
95
  "step": 55
96
  },
97
  {
98
- "epoch": 0.02,
99
- "grad_norm": 1957.0230712890625,
100
- "learning_rate": 0.0005,
101
- "loss": 20.642401123046874,
102
  "step": 60
103
  },
104
  {
105
- "epoch": 0.021666666666666667,
106
- "grad_norm": 238.84210205078125,
107
- "learning_rate": 0.0005,
108
- "loss": 21.76719970703125,
109
  "step": 65
110
  },
111
  {
112
- "epoch": 0.023333333333333334,
113
- "grad_norm": 71.29987335205078,
114
- "learning_rate": 0.0005,
115
- "loss": 20.405929565429688,
116
  "step": 70
117
  },
118
  {
119
- "epoch": 0.025,
120
- "grad_norm": 105.72884368896484,
121
- "learning_rate": 0.0005,
122
- "loss": 19.047821044921875,
123
  "step": 75
124
  },
125
  {
126
- "epoch": 0.02666666666666667,
127
- "grad_norm": 7008.24560546875,
128
- "learning_rate": 0.0005,
129
- "loss": 18.463946533203124,
130
  "step": 80
131
  },
132
  {
133
- "epoch": 0.028333333333333332,
134
- "grad_norm": 171495.171875,
135
- "learning_rate": 0.0005,
136
- "loss": 18.29894561767578,
137
  "step": 85
138
  },
139
  {
140
- "epoch": 0.03,
141
- "grad_norm": 122140.40625,
142
- "learning_rate": 0.0005,
143
- "loss": 18.62984619140625,
144
  "step": 90
145
  },
146
  {
147
- "epoch": 0.03166666666666667,
148
- "grad_norm": 8799.4404296875,
149
- "learning_rate": 0.0005,
150
- "loss": 18.06275177001953,
151
  "step": 95
152
  },
153
  {
154
- "epoch": 0.03333333333333333,
155
- "grad_norm": 435225.90625,
156
- "learning_rate": 0.0005,
157
- "loss": 17.608197021484376,
158
  "step": 100
159
  },
160
  {
161
- "epoch": 0.03333333333333333,
162
- "eval_loss": 8.762983322143555,
163
- "eval_runtime": 218.6047,
164
- "eval_samples_per_second": 1.363,
165
- "eval_steps_per_second": 0.682,
166
  "step": 100
167
  },
168
  {
169
- "epoch": 0.035,
170
- "grad_norm": 10736.8583984375,
171
- "learning_rate": 0.0005,
172
- "loss": 17.602194213867186,
173
  "step": 105
174
  },
175
  {
176
- "epoch": 0.03666666666666667,
177
- "grad_norm": 2109406.5,
178
- "learning_rate": 0.0005,
179
- "loss": 18.206829833984376,
180
  "step": 110
181
  },
182
  {
183
- "epoch": 0.03833333333333333,
184
- "grad_norm": 311519.9375,
185
- "learning_rate": 0.0005,
186
- "loss": 18.991905212402344,
187
  "step": 115
188
  },
189
  {
190
- "epoch": 0.04,
191
- "grad_norm": 295.97235107421875,
192
- "learning_rate": 0.0005,
193
- "loss": 19.67296447753906,
194
  "step": 120
195
  },
196
  {
197
- "epoch": 0.041666666666666664,
198
- "grad_norm": 31418684.0,
199
- "learning_rate": 0.0005,
200
- "loss": 20.369107055664063,
201
  "step": 125
202
  },
203
  {
204
- "epoch": 0.043333333333333335,
205
- "grad_norm": 3918581.0,
206
- "learning_rate": 0.0005,
207
- "loss": 20.751130676269533,
208
  "step": 130
209
  },
210
  {
211
- "epoch": 0.045,
212
- "grad_norm": 13102.0244140625,
213
- "learning_rate": 0.0005,
214
- "loss": 21.04936828613281,
215
  "step": 135
216
  },
217
  {
218
- "epoch": 0.04666666666666667,
219
- "grad_norm": 16994096.0,
220
- "learning_rate": 0.0005,
221
- "loss": 21.621359252929686,
222
  "step": 140
223
  },
224
  {
225
- "epoch": 0.04833333333333333,
226
- "grad_norm": 2272.0400390625,
227
- "learning_rate": 0.0005,
228
- "loss": 22.08568878173828,
229
  "step": 145
230
  },
231
  {
232
- "epoch": 0.05,
233
- "grad_norm": 4981.4296875,
234
- "learning_rate": 0.0005,
235
- "loss": 22.322303771972656,
236
  "step": 150
237
  },
238
  {
239
- "epoch": 0.05,
240
- "eval_loss": 11.168928146362305,
241
- "eval_runtime": 217.379,
242
- "eval_samples_per_second": 1.371,
243
- "eval_steps_per_second": 0.685,
244
  "step": 150
245
  },
246
  {
247
- "epoch": 0.051666666666666666,
248
- "grad_norm": 2926.708251953125,
249
- "learning_rate": 0.0005,
250
- "loss": 22.440846252441407,
251
  "step": 155
252
  },
253
  {
254
- "epoch": 0.05333333333333334,
255
- "grad_norm": 37812.78515625,
256
- "learning_rate": 0.0005,
257
- "loss": 22.036326599121093,
258
  "step": 160
259
  },
260
  {
261
- "epoch": 0.055,
262
- "grad_norm": 638.6507568359375,
263
- "learning_rate": 0.0005,
264
- "loss": 20.635609436035157,
265
  "step": 165
266
  },
267
  {
268
- "epoch": 0.056666666666666664,
269
- "grad_norm": 45.453582763671875,
270
- "learning_rate": 0.0005,
271
- "loss": 19.9438720703125,
272
  "step": 170
273
  },
274
  {
275
- "epoch": 0.058333333333333334,
276
- "grad_norm": 122177.8359375,
277
- "learning_rate": 0.0005,
278
- "loss": 19.473664855957033,
279
  "step": 175
280
  },
281
  {
282
- "epoch": 0.06,
283
- "grad_norm": 1840.748779296875,
284
- "learning_rate": 0.0005,
285
- "loss": 19.456393432617187,
286
  "step": 180
287
  },
288
  {
289
- "epoch": 0.06166666666666667,
290
- "grad_norm": 130342.7265625,
291
- "learning_rate": 0.0005,
292
- "loss": 19.97138366699219,
293
  "step": 185
294
  },
295
  {
296
- "epoch": 0.06333333333333334,
297
- "grad_norm": 26071.556640625,
298
- "learning_rate": 0.0005,
299
- "loss": 20.022232055664062,
300
  "step": 190
301
  },
302
  {
303
- "epoch": 0.065,
304
- "grad_norm": 6111.51513671875,
305
- "learning_rate": 0.0005,
306
- "loss": 20.719598388671876,
307
  "step": 195
308
  },
309
  {
310
- "epoch": 0.06666666666666667,
311
- "grad_norm": 387584.5,
312
- "learning_rate": 0.0005,
313
- "loss": 21.63736572265625,
314
  "step": 200
315
  },
316
  {
317
- "epoch": 0.06666666666666667,
318
- "eval_loss": 11.059845924377441,
319
- "eval_runtime": 216.5693,
320
- "eval_samples_per_second": 1.376,
321
- "eval_steps_per_second": 0.688,
322
  "step": 200
323
  }
324
  ],
325
  "logging_steps": 5,
326
- "max_steps": 3000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 9223372036854775807,
329
  "save_steps": 200,
@@ -339,7 +339,7 @@
339
  "attributes": {}
340
  }
341
  },
342
- "total_flos": 2276052487372800.0,
343
  "train_batch_size": 2,
344
  "trial_name": null,
345
  "trial_params": null
 
2
  "best_global_step": null,
3
  "best_metric": null,
4
  "best_model_checkpoint": null,
5
+ "epoch": 0.2,
6
  "eval_steps": 50,
7
  "global_step": 200,
8
  "is_hyper_param_search": false,
 
10
  "is_world_process_zero": true,
11
  "log_history": [
12
  {
13
+ "epoch": 0.005,
14
+ "grad_norm": 68.2828140258789,
15
+ "learning_rate": 3.2000000000000005e-05,
16
+ "loss": 3.547993850708008,
17
  "step": 5
18
  },
19
  {
20
+ "epoch": 0.01,
21
+ "grad_norm": 381.42022705078125,
22
+ "learning_rate": 7.2e-05,
23
+ "loss": 3.2959220886230467,
24
  "step": 10
25
  },
26
  {
27
+ "epoch": 0.015,
28
+ "grad_norm": 19640.955078125,
29
+ "learning_rate": 0.00011200000000000001,
30
+ "loss": 3.369282531738281,
31
  "step": 15
32
  },
33
  {
34
+ "epoch": 0.02,
35
+ "grad_norm": 205.24681091308594,
36
+ "learning_rate": 0.000152,
37
+ "loss": 3.256881332397461,
38
  "step": 20
39
  },
40
  {
41
+ "epoch": 0.025,
42
+ "grad_norm": 259.6229248046875,
43
+ "learning_rate": 0.000192,
44
+ "loss": 3.3879322052001952,
45
  "step": 25
46
  },
47
  {
48
+ "epoch": 0.03,
49
+ "grad_norm": 2096.502685546875,
50
+ "learning_rate": 0.0001999916943334945,
51
+ "loss": 3.4791305541992186,
52
  "step": 30
53
  },
54
  {
55
+ "epoch": 0.035,
56
+ "grad_norm": 29.51216697692871,
57
+ "learning_rate": 0.0001999579549278937,
58
+ "loss": 3.5544124603271485,
59
  "step": 35
60
  },
61
  {
62
+ "epoch": 0.04,
63
+ "grad_norm": 2868.5859375,
64
+ "learning_rate": 0.00019989827142936862,
65
+ "loss": 3.4086448669433596,
66
  "step": 40
67
  },
68
  {
69
+ "epoch": 0.045,
70
+ "grad_norm": 116.66321563720703,
71
+ "learning_rate": 0.00019981265932877488,
72
+ "loss": 3.26219482421875,
73
  "step": 45
74
  },
75
  {
76
+ "epoch": 0.05,
77
+ "grad_norm": 1199.3328857421875,
78
+ "learning_rate": 0.00019970114084673796,
79
+ "loss": 3.556040954589844,
80
  "step": 50
81
  },
82
  {
83
+ "epoch": 0.05,
84
+ "eval_loss": 3.7243571281433105,
85
+ "eval_runtime": 200.7945,
86
+ "eval_samples_per_second": 1.484,
87
+ "eval_steps_per_second": 0.299,
88
  "step": 50
89
  },
90
  {
91
+ "epoch": 0.055,
92
+ "grad_norm": 1653.0960693359375,
93
+ "learning_rate": 0.0001995637449278864,
94
+ "loss": 3.860651397705078,
95
  "step": 55
96
  },
97
  {
98
+ "epoch": 0.06,
99
+ "grad_norm": 411.4178466796875,
100
+ "learning_rate": 0.00019940050723333866,
101
+ "loss": 3.699799728393555,
102
  "step": 60
103
  },
104
  {
105
+ "epoch": 0.065,
106
+ "grad_norm": 2424.376708984375,
107
+ "learning_rate": 0.0001992114701314478,
108
+ "loss": 3.703348159790039,
109
  "step": 65
110
  },
111
  {
112
+ "epoch": 0.07,
113
+ "grad_norm": 23206.84375,
114
+ "learning_rate": 0.0001989966826868044,
115
+ "loss": 3.6570384979248045,
116
  "step": 70
117
  },
118
  {
119
+ "epoch": 0.075,
120
+ "grad_norm": 395.9013977050781,
121
+ "learning_rate": 0.00019875620064750202,
122
+ "loss": 3.622447204589844,
123
  "step": 75
124
  },
125
  {
126
+ "epoch": 0.08,
127
+ "grad_norm": 228293.453125,
128
+ "learning_rate": 0.00019849008643066772,
129
+ "loss": 3.773283767700195,
130
  "step": 80
131
  },
132
  {
133
+ "epoch": 0.085,
134
+ "grad_norm": 52893.1875,
135
+ "learning_rate": 0.00019819840910626174,
136
+ "loss": 3.650778961181641,
137
  "step": 85
138
  },
139
  {
140
+ "epoch": 0.09,
141
+ "grad_norm": 12679.291015625,
142
+ "learning_rate": 0.0001978812443791503,
143
+ "loss": 3.9305484771728514,
144
  "step": 90
145
  },
146
  {
147
+ "epoch": 0.095,
148
+ "grad_norm": 439037.4375,
149
+ "learning_rate": 0.0001975386745694565,
150
+ "loss": 3.8479537963867188,
151
  "step": 95
152
  },
153
  {
154
+ "epoch": 0.1,
155
+ "grad_norm": 878.3585815429688,
156
+ "learning_rate": 0.0001971707885911941,
157
+ "loss": 3.90667724609375,
158
  "step": 100
159
  },
160
  {
161
+ "epoch": 0.1,
162
+ "eval_loss": 4.1160054206848145,
163
+ "eval_runtime": 201.0129,
164
+ "eval_samples_per_second": 1.482,
165
+ "eval_steps_per_second": 0.298,
166
  "step": 100
167
  },
168
  {
169
+ "epoch": 0.105,
170
+ "grad_norm": 2385.37744140625,
171
+ "learning_rate": 0.00019677768192918971,
172
+ "loss": 3.805155944824219,
173
  "step": 105
174
  },
175
  {
176
+ "epoch": 0.11,
177
+ "grad_norm": 10.20463752746582,
178
+ "learning_rate": 0.00019635945661430006,
179
+ "loss": 3.822935104370117,
180
  "step": 110
181
  },
182
  {
183
+ "epoch": 0.115,
184
+ "grad_norm": 7.702394962310791,
185
+ "learning_rate": 0.0001959162211969295,
186
+ "loss": 3.702631378173828,
187
  "step": 115
188
  },
189
  {
190
+ "epoch": 0.12,
191
+ "grad_norm": 2.8037490844726562,
192
+ "learning_rate": 0.00019544809071885604,
193
+ "loss": 3.6845611572265624,
194
  "step": 120
195
  },
196
  {
197
+ "epoch": 0.125,
198
+ "grad_norm": 3.6743063926696777,
199
+ "learning_rate": 0.00019495518668337201,
200
+ "loss": 3.644290542602539,
201
  "step": 125
202
  },
203
  {
204
+ "epoch": 0.13,
205
+ "grad_norm": 5.9780964851379395,
206
+ "learning_rate": 0.00019443763702374812,
207
+ "loss": 3.5586227416992187,
208
  "step": 130
209
  },
210
  {
211
+ "epoch": 0.135,
212
+ "grad_norm": 2.230492353439331,
213
+ "learning_rate": 0.00019389557607002805,
214
+ "loss": 3.4619365692138673,
215
  "step": 135
216
  },
217
  {
218
+ "epoch": 0.14,
219
+ "grad_norm": 1.7121639251708984,
220
+ "learning_rate": 0.00019332914451416347,
221
+ "loss": 3.531117248535156,
222
  "step": 140
223
  },
224
  {
225
+ "epoch": 0.145,
226
+ "grad_norm": 4.843437194824219,
227
+ "learning_rate": 0.0001927384893734971,
228
+ "loss": 3.5250553131103515,
229
  "step": 145
230
  },
231
  {
232
+ "epoch": 0.15,
233
+ "grad_norm": 1.5022116899490356,
234
+ "learning_rate": 0.00019212376395260448,
235
+ "loss": 3.252106475830078,
236
  "step": 150
237
  },
238
  {
239
+ "epoch": 0.15,
240
+ "eval_loss": 3.604867935180664,
241
+ "eval_runtime": 200.7401,
242
+ "eval_samples_per_second": 1.485,
243
+ "eval_steps_per_second": 0.299,
244
  "step": 150
245
  },
246
  {
247
+ "epoch": 0.155,
248
+ "grad_norm": 2.400775909423828,
249
+ "learning_rate": 0.00019148512780350384,
250
+ "loss": 3.291422653198242,
251
  "step": 155
252
  },
253
  {
254
+ "epoch": 0.16,
255
+ "grad_norm": 2.1534276008605957,
256
+ "learning_rate": 0.00019082274668424422,
257
+ "loss": 3.4382129669189454,
258
  "step": 160
259
  },
260
  {
261
+ "epoch": 0.165,
262
+ "grad_norm": 5.130342960357666,
263
+ "learning_rate": 0.00019013679251588303,
264
+ "loss": 3.4447181701660154,
265
  "step": 165
266
  },
267
  {
268
+ "epoch": 0.17,
269
+ "grad_norm": 125.57601928710938,
270
+ "learning_rate": 0.00018942744333786397,
271
+ "loss": 3.4935020446777343,
272
  "step": 170
273
  },
274
  {
275
+ "epoch": 0.175,
276
+ "grad_norm": 7.2148356437683105,
277
+ "learning_rate": 0.00018869488326180679,
278
+ "loss": 3.4430862426757813,
279
  "step": 175
280
  },
281
  {
282
+ "epoch": 0.18,
283
+ "grad_norm": 47.17856979370117,
284
+ "learning_rate": 0.0001879393024237212,
285
+ "loss": 3.4849868774414063,
286
  "step": 180
287
  },
288
  {
289
+ "epoch": 0.185,
290
+ "grad_norm": 3.984619140625,
291
+ "learning_rate": 0.00018716089693465696,
292
+ "loss": 3.467825698852539,
293
  "step": 185
294
  },
295
  {
296
+ "epoch": 0.19,
297
+ "grad_norm": 1.7979289293289185,
298
+ "learning_rate": 0.00018635986882980325,
299
+ "loss": 3.2273773193359374,
300
  "step": 190
301
  },
302
  {
303
+ "epoch": 0.195,
304
+ "grad_norm": 49.847412109375,
305
+ "learning_rate": 0.00018553642601605068,
306
+ "loss": 3.51593017578125,
307
  "step": 195
308
  },
309
  {
310
+ "epoch": 0.2,
311
+ "grad_norm": 1.552674412727356,
312
+ "learning_rate": 0.0001846907822180286,
313
+ "loss": 3.5006553649902346,
314
  "step": 200
315
  },
316
  {
317
+ "epoch": 0.2,
318
+ "eval_loss": 3.619206190109253,
319
+ "eval_runtime": 201.0123,
320
+ "eval_samples_per_second": 1.482,
321
+ "eval_steps_per_second": 0.298,
322
  "step": 200
323
  }
324
  ],
325
  "logging_steps": 5,
326
+ "max_steps": 1000,
327
  "num_input_tokens_seen": 0,
328
  "num_train_epochs": 9223372036854775807,
329
  "save_steps": 200,
 
339
  "attributes": {}
340
  }
341
  },
342
+ "total_flos": 952423258521600.0,
343
  "train_batch_size": 2,
344
  "trial_name": null,
345
  "trial_params": null
last-checkpoint/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:56d4930d2f2e2d96034931eacf623f5cc07c816a8f7e9190dfff6ba278e2fb60
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:832e2094ad3db4c13c39e196610b6d644659d4779665163c375adaff3acee4e4
3
  size 5201