SpireLab commited on
Commit
d377330
·
verified ·
1 Parent(s): b9f7189

Delete train_data/muril_bh_domain/checkpoint-4000/trainer_state.json with huggingface_hub

Browse files
train_data/muril_bh_domain/checkpoint-4000/trainer_state.json DELETED
@@ -1,341 +0,0 @@
1
- {
2
- "best_metric": null,
3
- "best_model_checkpoint": null,
4
- "epoch": 2.530844669408415,
5
- "eval_steps": 1000,
6
- "global_step": 4000,
7
- "is_hyper_param_search": false,
8
- "is_local_process_zero": true,
9
- "is_world_process_zero": true,
10
- "log_history": [
11
- {
12
- "epoch": 0.06327111673521038,
13
- "grad_norm": 4.5707688331604,
14
- "learning_rate": 1.0548523206751056e-05,
15
- "loss": 6.0536,
16
- "step": 100
17
- },
18
- {
19
- "epoch": 0.12654223347042076,
20
- "grad_norm": 7.803924083709717,
21
- "learning_rate": 2.1097046413502112e-05,
22
- "loss": 5.8177,
23
- "step": 200
24
- },
25
- {
26
- "epoch": 0.18981335020563114,
27
- "grad_norm": 8.050126075744629,
28
- "learning_rate": 3.1645569620253167e-05,
29
- "loss": 5.0311,
30
- "step": 300
31
- },
32
- {
33
- "epoch": 0.2530844669408415,
34
- "grad_norm": 8.803975105285645,
35
- "learning_rate": 4.2194092827004224e-05,
36
- "loss": 4.7351,
37
- "step": 400
38
- },
39
- {
40
- "epoch": 0.3163555836760519,
41
- "grad_norm": 12.846040725708008,
42
- "learning_rate": 4.96952648851383e-05,
43
- "loss": 4.5453,
44
- "step": 500
45
- },
46
- {
47
- "epoch": 0.3796267004112623,
48
- "grad_norm": 15.128095626831055,
49
- "learning_rate": 4.852320675105486e-05,
50
- "loss": 4.3979,
51
- "step": 600
52
- },
53
- {
54
- "epoch": 0.44289781714647264,
55
- "grad_norm": 10.33956527709961,
56
- "learning_rate": 4.7351148616971405e-05,
57
- "loss": 4.4056,
58
- "step": 700
59
- },
60
- {
61
- "epoch": 0.506168933881683,
62
- "grad_norm": 23.287179946899414,
63
- "learning_rate": 4.617909048288795e-05,
64
- "loss": 4.2787,
65
- "step": 800
66
- },
67
- {
68
- "epoch": 0.5694400506168934,
69
- "grad_norm": 12.955565452575684,
70
- "learning_rate": 4.50070323488045e-05,
71
- "loss": 4.3504,
72
- "step": 900
73
- },
74
- {
75
- "epoch": 0.6327111673521038,
76
- "grad_norm": 10.07479476928711,
77
- "learning_rate": 4.3834974214721055e-05,
78
- "loss": 4.2521,
79
- "step": 1000
80
- },
81
- {
82
- "epoch": 0.6327111673521038,
83
- "eval_runtime": 33.0567,
84
- "eval_samples_per_second": 95.593,
85
- "eval_steps_per_second": 11.949,
86
- "step": 1000
87
- },
88
- {
89
- "epoch": 0.6959822840873141,
90
- "grad_norm": 12.32780647277832,
91
- "learning_rate": 4.26629160806376e-05,
92
- "loss": 4.2314,
93
- "step": 1100
94
- },
95
- {
96
- "epoch": 0.7592534008225246,
97
- "grad_norm": 15.515801429748535,
98
- "learning_rate": 4.149085794655415e-05,
99
- "loss": 4.3058,
100
- "step": 1200
101
- },
102
- {
103
- "epoch": 0.8225245175577349,
104
- "grad_norm": 12.472834587097168,
105
- "learning_rate": 4.03187998124707e-05,
106
- "loss": 4.2158,
107
- "step": 1300
108
- },
109
- {
110
- "epoch": 0.8857956342929453,
111
- "grad_norm": 16.903112411499023,
112
- "learning_rate": 3.914674167838725e-05,
113
- "loss": 4.142,
114
- "step": 1400
115
- },
116
- {
117
- "epoch": 0.9490667510281556,
118
- "grad_norm": 13.487791061401367,
119
- "learning_rate": 3.79746835443038e-05,
120
- "loss": 4.0534,
121
- "step": 1500
122
- },
123
- {
124
- "epoch": 1.012337867763366,
125
- "grad_norm": 14.721494674682617,
126
- "learning_rate": 3.680262541022035e-05,
127
- "loss": 4.1405,
128
- "step": 1600
129
- },
130
- {
131
- "epoch": 1.0756089844985763,
132
- "grad_norm": 16.011690139770508,
133
- "learning_rate": 3.56305672761369e-05,
134
- "loss": 4.1192,
135
- "step": 1700
136
- },
137
- {
138
- "epoch": 1.1388801012337868,
139
- "grad_norm": 15.692846298217773,
140
- "learning_rate": 3.445850914205345e-05,
141
- "loss": 4.1407,
142
- "step": 1800
143
- },
144
- {
145
- "epoch": 1.2021512179689973,
146
- "grad_norm": 13.71811294555664,
147
- "learning_rate": 3.328645100797e-05,
148
- "loss": 4.1839,
149
- "step": 1900
150
- },
151
- {
152
- "epoch": 1.2654223347042075,
153
- "grad_norm": 13.474448204040527,
154
- "learning_rate": 3.2114392873886545e-05,
155
- "loss": 4.0813,
156
- "step": 2000
157
- },
158
- {
159
- "epoch": 1.2654223347042075,
160
- "eval_runtime": 33.0889,
161
- "eval_samples_per_second": 95.5,
162
- "eval_steps_per_second": 11.938,
163
- "step": 2000
164
- },
165
- {
166
- "epoch": 1.328693451439418,
167
- "grad_norm": 15.276654243469238,
168
- "learning_rate": 3.09423347398031e-05,
169
- "loss": 3.9624,
170
- "step": 2100
171
- },
172
- {
173
- "epoch": 1.3919645681746282,
174
- "grad_norm": 13.796238899230957,
175
- "learning_rate": 2.9770276605719643e-05,
176
- "loss": 4.0653,
177
- "step": 2200
178
- },
179
- {
180
- "epoch": 1.4552356849098387,
181
- "grad_norm": 16.452594757080078,
182
- "learning_rate": 2.8598218471636194e-05,
183
- "loss": 4.0338,
184
- "step": 2300
185
- },
186
- {
187
- "epoch": 1.518506801645049,
188
- "grad_norm": 20.27753257751465,
189
- "learning_rate": 2.7426160337552742e-05,
190
- "loss": 4.0962,
191
- "step": 2400
192
- },
193
- {
194
- "epoch": 1.5817779183802594,
195
- "grad_norm": 15.492554664611816,
196
- "learning_rate": 2.6254102203469293e-05,
197
- "loss": 4.057,
198
- "step": 2500
199
- },
200
- {
201
- "epoch": 1.6450490351154698,
202
- "grad_norm": 15.298516273498535,
203
- "learning_rate": 2.508204406938584e-05,
204
- "loss": 4.0301,
205
- "step": 2600
206
- },
207
- {
208
- "epoch": 1.70832015185068,
209
- "grad_norm": 15.670785903930664,
210
- "learning_rate": 2.3909985935302392e-05,
211
- "loss": 4.0129,
212
- "step": 2700
213
- },
214
- {
215
- "epoch": 1.7715912685858906,
216
- "grad_norm": 18.541555404663086,
217
- "learning_rate": 2.2737927801218943e-05,
218
- "loss": 3.9724,
219
- "step": 2800
220
- },
221
- {
222
- "epoch": 1.834862385321101,
223
- "grad_norm": 19.13411521911621,
224
- "learning_rate": 2.156586966713549e-05,
225
- "loss": 4.0044,
226
- "step": 2900
227
- },
228
- {
229
- "epoch": 1.8981335020563113,
230
- "grad_norm": 14.532624244689941,
231
- "learning_rate": 2.039381153305204e-05,
232
- "loss": 3.9882,
233
- "step": 3000
234
- },
235
- {
236
- "epoch": 1.8981335020563113,
237
- "eval_runtime": 33.1181,
238
- "eval_samples_per_second": 95.416,
239
- "eval_steps_per_second": 11.927,
240
- "step": 3000
241
- },
242
- {
243
- "epoch": 1.9614046187915217,
244
- "grad_norm": 15.767202377319336,
245
- "learning_rate": 1.922175339896859e-05,
246
- "loss": 3.9372,
247
- "step": 3100
248
- },
249
- {
250
- "epoch": 2.024675735526732,
251
- "grad_norm": 17.210546493530273,
252
- "learning_rate": 1.804969526488514e-05,
253
- "loss": 3.9757,
254
- "step": 3200
255
- },
256
- {
257
- "epoch": 2.0879468522619424,
258
- "grad_norm": 15.209254264831543,
259
- "learning_rate": 1.6877637130801688e-05,
260
- "loss": 3.9668,
261
- "step": 3300
262
- },
263
- {
264
- "epoch": 2.1512179689971527,
265
- "grad_norm": 16.821176528930664,
266
- "learning_rate": 1.570557899671824e-05,
267
- "loss": 3.9732,
268
- "step": 3400
269
- },
270
- {
271
- "epoch": 2.2144890857323634,
272
- "grad_norm": 15.914960861206055,
273
- "learning_rate": 1.4533520862634786e-05,
274
- "loss": 3.9375,
275
- "step": 3500
276
- },
277
- {
278
- "epoch": 2.2777602024675736,
279
- "grad_norm": 16.489627838134766,
280
- "learning_rate": 1.3361462728551336e-05,
281
- "loss": 3.9993,
282
- "step": 3600
283
- },
284
- {
285
- "epoch": 2.341031319202784,
286
- "grad_norm": 17.943021774291992,
287
- "learning_rate": 1.2189404594467887e-05,
288
- "loss": 3.9889,
289
- "step": 3700
290
- },
291
- {
292
- "epoch": 2.4043024359379945,
293
- "grad_norm": 14.150239944458008,
294
- "learning_rate": 1.1017346460384436e-05,
295
- "loss": 4.0,
296
- "step": 3800
297
- },
298
- {
299
- "epoch": 2.4675735526732048,
300
- "grad_norm": 15.843707084655762,
301
- "learning_rate": 9.845288326300985e-06,
302
- "loss": 3.9527,
303
- "step": 3900
304
- },
305
- {
306
- "epoch": 2.530844669408415,
307
- "grad_norm": 16.922142028808594,
308
- "learning_rate": 8.673230192217533e-06,
309
- "loss": 3.8801,
310
- "step": 4000
311
- },
312
- {
313
- "epoch": 2.530844669408415,
314
- "eval_runtime": 33.0326,
315
- "eval_samples_per_second": 95.663,
316
- "eval_steps_per_second": 11.958,
317
- "step": 4000
318
- }
319
- ],
320
- "logging_steps": 100,
321
- "max_steps": 4740,
322
- "num_input_tokens_seen": 0,
323
- "num_train_epochs": 3,
324
- "save_steps": 1000,
325
- "stateful_callbacks": {
326
- "TrainerControl": {
327
- "args": {
328
- "should_epoch_stop": false,
329
- "should_evaluate": false,
330
- "should_log": false,
331
- "should_save": true,
332
- "should_training_stop": false
333
- },
334
- "attributes": {}
335
- }
336
- },
337
- "total_flos": 1700301256769430.0,
338
- "train_batch_size": 8,
339
- "trial_name": null,
340
- "trial_params": null
341
- }