PedramR commited on
Commit
ffdcc64
·
verified ·
1 Parent(s): c9216d2

checkpoint-1500

Browse files
qwen3.5-2b/checkpoint-100/trainer_state.json DELETED
@@ -1,112 +0,0 @@
1
- {
2
- "best_global_step": null,
3
- "best_metric": null,
4
- "best_model_checkpoint": null,
5
- "epoch": 0.07914523149980214,
6
- "eval_steps": 100,
7
- "global_step": 100,
8
- "is_hyper_param_search": false,
9
- "is_local_process_zero": true,
10
- "is_world_process_zero": true,
11
- "log_history": [
12
- {
13
- "epoch": 0.007914523149980214,
14
- "grad_norm": 1.110097050666809,
15
- "learning_rate": 1.4173228346456694e-05,
16
- "loss": 0.9039163589477539,
17
- "step": 10
18
- },
19
- {
20
- "epoch": 0.015829046299960427,
21
- "grad_norm": 0.8214025497436523,
22
- "learning_rate": 2.992125984251969e-05,
23
- "loss": 0.7784994125366211,
24
- "step": 20
25
- },
26
- {
27
- "epoch": 0.02374356944994064,
28
- "grad_norm": 0.796338677406311,
29
- "learning_rate": 4.566929133858268e-05,
30
- "loss": 0.694743537902832,
31
- "step": 30
32
- },
33
- {
34
- "epoch": 0.031658092599920855,
35
- "grad_norm": 0.7282607555389404,
36
- "learning_rate": 6.141732283464568e-05,
37
- "loss": 0.6986559391021728,
38
- "step": 40
39
- },
40
- {
41
- "epoch": 0.03957261574990107,
42
- "grad_norm": 1.00419020652771,
43
- "learning_rate": 7.716535433070867e-05,
44
- "loss": 0.697541332244873,
45
- "step": 50
46
- },
47
- {
48
- "epoch": 0.04748713889988128,
49
- "grad_norm": 0.7905988097190857,
50
- "learning_rate": 9.291338582677166e-05,
51
- "loss": 0.586260461807251,
52
- "step": 60
53
- },
54
- {
55
- "epoch": 0.055401662049861494,
56
- "grad_norm": 0.8630598187446594,
57
- "learning_rate": 0.00010866141732283466,
58
- "loss": 0.6937695503234863,
59
- "step": 70
60
- },
61
- {
62
- "epoch": 0.06331618519984171,
63
- "grad_norm": 0.7757033705711365,
64
- "learning_rate": 0.00012440944881889765,
65
- "loss": 0.652254295349121,
66
- "step": 80
67
- },
68
- {
69
- "epoch": 0.07123070834982193,
70
- "grad_norm": 0.8161948919296265,
71
- "learning_rate": 0.00014015748031496062,
72
- "loss": 0.6520013332366943,
73
- "step": 90
74
- },
75
- {
76
- "epoch": 0.07914523149980214,
77
- "grad_norm": 0.8955326676368713,
78
- "learning_rate": 0.00015590551181102362,
79
- "loss": 0.6528121948242187,
80
- "step": 100
81
- },
82
- {
83
- "epoch": 0.07914523149980214,
84
- "eval_loss": 0.6273276209831238,
85
- "eval_runtime": 460.2352,
86
- "eval_samples_per_second": 24.338,
87
- "eval_steps_per_second": 6.086,
88
- "step": 100
89
- }
90
- ],
91
- "logging_steps": 10,
92
- "max_steps": 2528,
93
- "num_input_tokens_seen": 0,
94
- "num_train_epochs": 2,
95
- "save_steps": 100,
96
- "stateful_callbacks": {
97
- "TrainerControl": {
98
- "args": {
99
- "should_epoch_stop": false,
100
- "should_evaluate": false,
101
- "should_log": false,
102
- "should_save": true,
103
- "should_training_stop": false
104
- },
105
- "attributes": {}
106
- }
107
- },
108
- "total_flos": 1.5094044096262656e+16,
109
- "train_batch_size": 4,
110
- "trial_name": null,
111
- "trial_params": null
112
- }
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/README.md RENAMED
File without changes
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/adapter_config.json RENAMED
@@ -32,18 +32,18 @@
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
 
 
35
  "gate_proj",
36
- "in_proj_b",
37
  "out_proj",
38
- "k_proj",
39
- "o_proj",
40
- "q_proj",
41
  "in_proj_z",
 
 
42
  "v_proj",
43
- "up_proj",
44
- "in_proj_qkv",
45
- "down_proj",
46
- "in_proj_a"
47
  ],
48
  "target_parameters": null,
49
  "task_type": "CAUSAL_LM",
 
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
35
+ "up_proj",
36
+ "down_proj",
37
  "gate_proj",
 
38
  "out_proj",
39
+ "in_proj_a",
40
+ "in_proj_qkv",
 
41
  "in_proj_z",
42
+ "o_proj",
43
+ "in_proj_b",
44
  "v_proj",
45
+ "k_proj",
46
+ "q_proj"
 
 
47
  ],
48
  "target_parameters": null,
49
  "task_type": "CAUSAL_LM",
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b354598c9fedef3d5218417ffbf82c8ae665cf3c58e8ddbe8d685b8ddd06bf4e
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bec024ade9ae0201f56b2f969e705fc98ef28b2b12db40a7b0e0f63c32eca119
3
  size 134604152
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4c25f3a565b4cb2dc34417c2eb57451e6c275e00d4a84d70916a3700bc38afaa
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b05fd1dcc583dcb2cb90241178cb6af30091739ac8c7c29651b243114a65e472
3
  size 269426431
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0bd3c954d401b65bf65ab4946a048a3a709cb2bbbe22c8401248fde3d9108ce2
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2cc7717946ad1ca0e68e937394df05d93f33d3ea424e24a821708701168d1829
3
  size 14645
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:810230b50d30862ac3f8ee7cf9a4fed23d27151927c9bd74658f2abfc3b2d798
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0a8c4e551c765686f575f320c58dfb3f769671fd883afc927f1c6e8a1b15d8ba
3
  size 1465
qwen3.5-2b/checkpoint-1500/trainer_state.json ADDED
@@ -0,0 +1,1124 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.791817881887166,
6
+ "eval_steps": 500,
7
+ "global_step": 1500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.005278785879247773,
14
+ "grad_norm": 1.305302381515503,
15
+ "learning_rate": 1.8947368421052634e-05,
16
+ "loss": 0.8562479972839355,
17
+ "step": 10
18
+ },
19
+ {
20
+ "epoch": 0.010557571758495546,
21
+ "grad_norm": 0.9230896830558777,
22
+ "learning_rate": 4e-05,
23
+ "loss": 0.7965596675872803,
24
+ "step": 20
25
+ },
26
+ {
27
+ "epoch": 0.01583635763774332,
28
+ "grad_norm": 0.7995359301567078,
29
+ "learning_rate": 6.105263157894737e-05,
30
+ "loss": 0.7992881298065185,
31
+ "step": 30
32
+ },
33
+ {
34
+ "epoch": 0.02111514351699109,
35
+ "grad_norm": 0.7340251207351685,
36
+ "learning_rate": 8.210526315789474e-05,
37
+ "loss": 0.6521625995635987,
38
+ "step": 40
39
+ },
40
+ {
41
+ "epoch": 0.026393929396238865,
42
+ "grad_norm": 1.0072991847991943,
43
+ "learning_rate": 0.00010315789473684211,
44
+ "loss": 0.6839987277984619,
45
+ "step": 50
46
+ },
47
+ {
48
+ "epoch": 0.03167271527548664,
49
+ "grad_norm": 0.8970349431037903,
50
+ "learning_rate": 0.00012421052631578949,
51
+ "loss": 0.6341384887695313,
52
+ "step": 60
53
+ },
54
+ {
55
+ "epoch": 0.03695150115473441,
56
+ "grad_norm": 0.883733332157135,
57
+ "learning_rate": 0.00014526315789473686,
58
+ "loss": 0.6603663444519043,
59
+ "step": 70
60
+ },
61
+ {
62
+ "epoch": 0.04223028703398218,
63
+ "grad_norm": 0.8134523630142212,
64
+ "learning_rate": 0.00016631578947368423,
65
+ "loss": 0.6564276695251465,
66
+ "step": 80
67
+ },
68
+ {
69
+ "epoch": 0.047509072913229956,
70
+ "grad_norm": 0.8890961408615112,
71
+ "learning_rate": 0.0001873684210526316,
72
+ "loss": 0.6585474967956543,
73
+ "step": 90
74
+ },
75
+ {
76
+ "epoch": 0.05278785879247773,
77
+ "grad_norm": 0.7649273872375488,
78
+ "learning_rate": 0.00019955555555555558,
79
+ "loss": 0.6272543430328369,
80
+ "step": 100
81
+ },
82
+ {
83
+ "epoch": 0.0580666446717255,
84
+ "grad_norm": 0.7554565072059631,
85
+ "learning_rate": 0.00019844444444444445,
86
+ "loss": 0.5757376194000244,
87
+ "step": 110
88
+ },
89
+ {
90
+ "epoch": 0.06334543055097328,
91
+ "grad_norm": 0.7954307794570923,
92
+ "learning_rate": 0.00019733333333333335,
93
+ "loss": 0.641472053527832,
94
+ "step": 120
95
+ },
96
+ {
97
+ "epoch": 0.06862421643022105,
98
+ "grad_norm": 0.672359824180603,
99
+ "learning_rate": 0.00019622222222222225,
100
+ "loss": 0.563064432144165,
101
+ "step": 130
102
+ },
103
+ {
104
+ "epoch": 0.07390300230946882,
105
+ "grad_norm": 0.6992788314819336,
106
+ "learning_rate": 0.0001951111111111111,
107
+ "loss": 0.600986099243164,
108
+ "step": 140
109
+ },
110
+ {
111
+ "epoch": 0.0791817881887166,
112
+ "grad_norm": 0.6328181028366089,
113
+ "learning_rate": 0.000194,
114
+ "loss": 0.5295759201049804,
115
+ "step": 150
116
+ },
117
+ {
118
+ "epoch": 0.08446057406796437,
119
+ "grad_norm": 0.6692398190498352,
120
+ "learning_rate": 0.0001928888888888889,
121
+ "loss": 0.6027700901031494,
122
+ "step": 160
123
+ },
124
+ {
125
+ "epoch": 0.08973935994721215,
126
+ "grad_norm": 0.7081276774406433,
127
+ "learning_rate": 0.0001917777777777778,
128
+ "loss": 0.5615420341491699,
129
+ "step": 170
130
+ },
131
+ {
132
+ "epoch": 0.09501814582645991,
133
+ "grad_norm": 0.5484495759010315,
134
+ "learning_rate": 0.00019066666666666668,
135
+ "loss": 0.5883731842041016,
136
+ "step": 180
137
+ },
138
+ {
139
+ "epoch": 0.10029693170570769,
140
+ "grad_norm": 0.7612242102622986,
141
+ "learning_rate": 0.00018955555555555558,
142
+ "loss": 0.5624749183654785,
143
+ "step": 190
144
+ },
145
+ {
146
+ "epoch": 0.10557571758495546,
147
+ "grad_norm": 0.678548276424408,
148
+ "learning_rate": 0.00018844444444444445,
149
+ "loss": 0.5546360969543457,
150
+ "step": 200
151
+ },
152
+ {
153
+ "epoch": 0.11085450346420324,
154
+ "grad_norm": 0.5756900906562805,
155
+ "learning_rate": 0.00018733333333333335,
156
+ "loss": 0.5600387573242187,
157
+ "step": 210
158
+ },
159
+ {
160
+ "epoch": 0.116133289343451,
161
+ "grad_norm": 0.5475565791130066,
162
+ "learning_rate": 0.00018622222222222223,
163
+ "loss": 0.5820147037506104,
164
+ "step": 220
165
+ },
166
+ {
167
+ "epoch": 0.12141207522269878,
168
+ "grad_norm": 0.6964915990829468,
169
+ "learning_rate": 0.00018511111111111113,
170
+ "loss": 0.5699102401733398,
171
+ "step": 230
172
+ },
173
+ {
174
+ "epoch": 0.12669086110194655,
175
+ "grad_norm": 0.6508777737617493,
176
+ "learning_rate": 0.00018400000000000003,
177
+ "loss": 0.5863839626312256,
178
+ "step": 240
179
+ },
180
+ {
181
+ "epoch": 0.13196964698119432,
182
+ "grad_norm": 0.6375535726547241,
183
+ "learning_rate": 0.00018288888888888887,
184
+ "loss": 0.5772487640380859,
185
+ "step": 250
186
+ },
187
+ {
188
+ "epoch": 0.1372484328604421,
189
+ "grad_norm": 0.5595227479934692,
190
+ "learning_rate": 0.00018177777777777778,
191
+ "loss": 0.603016710281372,
192
+ "step": 260
193
+ },
194
+ {
195
+ "epoch": 0.14252721873968988,
196
+ "grad_norm": 0.5612871646881104,
197
+ "learning_rate": 0.00018066666666666668,
198
+ "loss": 0.5080572605133057,
199
+ "step": 270
200
+ },
201
+ {
202
+ "epoch": 0.14780600461893764,
203
+ "grad_norm": 0.762787938117981,
204
+ "learning_rate": 0.00017955555555555558,
205
+ "loss": 0.6409445762634277,
206
+ "step": 280
207
+ },
208
+ {
209
+ "epoch": 0.1530847904981854,
210
+ "grad_norm": 0.6885519623756409,
211
+ "learning_rate": 0.00017844444444444445,
212
+ "loss": 0.6149398803710937,
213
+ "step": 290
214
+ },
215
+ {
216
+ "epoch": 0.1583635763774332,
217
+ "grad_norm": 0.6986534595489502,
218
+ "learning_rate": 0.00017733333333333335,
219
+ "loss": 0.5259696960449218,
220
+ "step": 300
221
+ },
222
+ {
223
+ "epoch": 0.16364236225668097,
224
+ "grad_norm": 0.601087212562561,
225
+ "learning_rate": 0.00017622222222222223,
226
+ "loss": 0.6464223384857177,
227
+ "step": 310
228
+ },
229
+ {
230
+ "epoch": 0.16892114813592873,
231
+ "grad_norm": 0.4432871639728546,
232
+ "learning_rate": 0.00017511111111111113,
233
+ "loss": 0.509718942642212,
234
+ "step": 320
235
+ },
236
+ {
237
+ "epoch": 0.1741999340151765,
238
+ "grad_norm": 0.5295445322990417,
239
+ "learning_rate": 0.000174,
240
+ "loss": 0.5282030582427979,
241
+ "step": 330
242
+ },
243
+ {
244
+ "epoch": 0.1794787198944243,
245
+ "grad_norm": 0.695189893245697,
246
+ "learning_rate": 0.0001728888888888889,
247
+ "loss": 0.5390444278717041,
248
+ "step": 340
249
+ },
250
+ {
251
+ "epoch": 0.18475750577367206,
252
+ "grad_norm": 0.6871212124824524,
253
+ "learning_rate": 0.0001717777777777778,
254
+ "loss": 0.6001670360565186,
255
+ "step": 350
256
+ },
257
+ {
258
+ "epoch": 0.19003629165291983,
259
+ "grad_norm": 0.5886061787605286,
260
+ "learning_rate": 0.00017066666666666668,
261
+ "loss": 0.5532281875610352,
262
+ "step": 360
263
+ },
264
+ {
265
+ "epoch": 0.1953150775321676,
266
+ "grad_norm": 0.4359692931175232,
267
+ "learning_rate": 0.00016955555555555555,
268
+ "loss": 0.5710611343383789,
269
+ "step": 370
270
+ },
271
+ {
272
+ "epoch": 0.20059386341141539,
273
+ "grad_norm": 0.5859516859054565,
274
+ "learning_rate": 0.00016844444444444445,
275
+ "loss": 0.527172040939331,
276
+ "step": 380
277
+ },
278
+ {
279
+ "epoch": 0.20587264929066315,
280
+ "grad_norm": 0.595167875289917,
281
+ "learning_rate": 0.00016733333333333335,
282
+ "loss": 0.5060821056365967,
283
+ "step": 390
284
+ },
285
+ {
286
+ "epoch": 0.21115143516991092,
287
+ "grad_norm": 0.5597789883613586,
288
+ "learning_rate": 0.00016622222222222223,
289
+ "loss": 0.5472925186157227,
290
+ "step": 400
291
+ },
292
+ {
293
+ "epoch": 0.21643022104915868,
294
+ "grad_norm": 0.5002998113632202,
295
+ "learning_rate": 0.00016511111111111113,
296
+ "loss": 0.5917861461639404,
297
+ "step": 410
298
+ },
299
+ {
300
+ "epoch": 0.22170900692840648,
301
+ "grad_norm": 0.6672388911247253,
302
+ "learning_rate": 0.000164,
303
+ "loss": 0.5691781044006348,
304
+ "step": 420
305
+ },
306
+ {
307
+ "epoch": 0.22698779280765424,
308
+ "grad_norm": 0.580162763595581,
309
+ "learning_rate": 0.0001628888888888889,
310
+ "loss": 0.5123159408569335,
311
+ "step": 430
312
+ },
313
+ {
314
+ "epoch": 0.232266578686902,
315
+ "grad_norm": 0.5568054914474487,
316
+ "learning_rate": 0.00016177777777777778,
317
+ "loss": 0.5455626010894775,
318
+ "step": 440
319
+ },
320
+ {
321
+ "epoch": 0.23754536456614977,
322
+ "grad_norm": 0.6651593446731567,
323
+ "learning_rate": 0.00016066666666666668,
324
+ "loss": 0.5399269104003906,
325
+ "step": 450
326
+ },
327
+ {
328
+ "epoch": 0.24282415044539757,
329
+ "grad_norm": 0.5001797080039978,
330
+ "learning_rate": 0.00015955555555555558,
331
+ "loss": 0.5277121543884278,
332
+ "step": 460
333
+ },
334
+ {
335
+ "epoch": 0.24810293632464533,
336
+ "grad_norm": 0.5514921545982361,
337
+ "learning_rate": 0.00015844444444444445,
338
+ "loss": 0.5342081546783447,
339
+ "step": 470
340
+ },
341
+ {
342
+ "epoch": 0.2533817222038931,
343
+ "grad_norm": 0.5599192380905151,
344
+ "learning_rate": 0.00015733333333333333,
345
+ "loss": 0.5150089263916016,
346
+ "step": 480
347
+ },
348
+ {
349
+ "epoch": 0.2586605080831409,
350
+ "grad_norm": 0.563194215297699,
351
+ "learning_rate": 0.00015622222222222223,
352
+ "loss": 0.5231950759887696,
353
+ "step": 490
354
+ },
355
+ {
356
+ "epoch": 0.26393929396238863,
357
+ "grad_norm": 0.7210673689842224,
358
+ "learning_rate": 0.00015511111111111113,
359
+ "loss": 0.5708573818206787,
360
+ "step": 500
361
+ },
362
+ {
363
+ "epoch": 0.26393929396238863,
364
+ "eval_loss": 0.534747838973999,
365
+ "eval_runtime": 459.3797,
366
+ "eval_samples_per_second": 24.383,
367
+ "eval_steps_per_second": 6.097,
368
+ "step": 500
369
+ },
370
+ {
371
+ "epoch": 0.26393929396238863,
372
+ "step": 500,
373
+ "test_best_wer": 0.21571711227233525,
374
+ "test_exact_match": 0.0,
375
+ "test_hallucination_rate": 0.09375,
376
+ "test_wer": 0.4919080716351276
377
+ },
378
+ {
379
+ "epoch": 0.2692180798416364,
380
+ "grad_norm": 0.5391792058944702,
381
+ "learning_rate": 0.000154,
382
+ "loss": 0.5354532241821289,
383
+ "step": 510
384
+ },
385
+ {
386
+ "epoch": 0.2744968657208842,
387
+ "grad_norm": 0.5327297449111938,
388
+ "learning_rate": 0.0001528888888888889,
389
+ "loss": 0.5177035331726074,
390
+ "step": 520
391
+ },
392
+ {
393
+ "epoch": 0.27977565160013196,
394
+ "grad_norm": 0.5588216781616211,
395
+ "learning_rate": 0.00015177777777777778,
396
+ "loss": 0.5256490707397461,
397
+ "step": 530
398
+ },
399
+ {
400
+ "epoch": 0.28505443747937975,
401
+ "grad_norm": 0.6024046540260315,
402
+ "learning_rate": 0.00015066666666666668,
403
+ "loss": 0.5359174728393554,
404
+ "step": 540
405
+ },
406
+ {
407
+ "epoch": 0.2903332233586275,
408
+ "grad_norm": 0.5452874898910522,
409
+ "learning_rate": 0.00014955555555555555,
410
+ "loss": 0.5050687313079834,
411
+ "step": 550
412
+ },
413
+ {
414
+ "epoch": 0.2956120092378753,
415
+ "grad_norm": 0.5716919898986816,
416
+ "learning_rate": 0.00014844444444444445,
417
+ "loss": 0.5043536186218261,
418
+ "step": 560
419
+ },
420
+ {
421
+ "epoch": 0.3008907951171231,
422
+ "grad_norm": 0.5621301531791687,
423
+ "learning_rate": 0.00014733333333333335,
424
+ "loss": 0.5681197643280029,
425
+ "step": 570
426
+ },
427
+ {
428
+ "epoch": 0.3061695809963708,
429
+ "grad_norm": 0.49813947081565857,
430
+ "learning_rate": 0.00014622222222222223,
431
+ "loss": 0.5073559761047364,
432
+ "step": 580
433
+ },
434
+ {
435
+ "epoch": 0.3114483668756186,
436
+ "grad_norm": 0.5197216272354126,
437
+ "learning_rate": 0.0001451111111111111,
438
+ "loss": 0.510869026184082,
439
+ "step": 590
440
+ },
441
+ {
442
+ "epoch": 0.3167271527548664,
443
+ "grad_norm": 0.5098935961723328,
444
+ "learning_rate": 0.000144,
445
+ "loss": 0.5177080154418945,
446
+ "step": 600
447
+ },
448
+ {
449
+ "epoch": 0.32200593863411414,
450
+ "grad_norm": 0.5004557967185974,
451
+ "learning_rate": 0.0001428888888888889,
452
+ "loss": 0.5292775154113769,
453
+ "step": 610
454
+ },
455
+ {
456
+ "epoch": 0.32728472451336194,
457
+ "grad_norm": 0.6179164052009583,
458
+ "learning_rate": 0.00014177777777777778,
459
+ "loss": 0.532755994796753,
460
+ "step": 620
461
+ },
462
+ {
463
+ "epoch": 0.3325635103926097,
464
+ "grad_norm": 0.4971306025981903,
465
+ "learning_rate": 0.00014066666666666668,
466
+ "loss": 0.5370781421661377,
467
+ "step": 630
468
+ },
469
+ {
470
+ "epoch": 0.33784229627185747,
471
+ "grad_norm": 0.5876296758651733,
472
+ "learning_rate": 0.00013955555555555558,
473
+ "loss": 0.5408426284790039,
474
+ "step": 640
475
+ },
476
+ {
477
+ "epoch": 0.34312108215110526,
478
+ "grad_norm": 0.4632679522037506,
479
+ "learning_rate": 0.00013844444444444445,
480
+ "loss": 0.475573205947876,
481
+ "step": 650
482
+ },
483
+ {
484
+ "epoch": 0.348399868030353,
485
+ "grad_norm": 0.6051201224327087,
486
+ "learning_rate": 0.00013733333333333333,
487
+ "loss": 0.5176257133483887,
488
+ "step": 660
489
+ },
490
+ {
491
+ "epoch": 0.3536786539096008,
492
+ "grad_norm": 0.5086127519607544,
493
+ "learning_rate": 0.00013622222222222223,
494
+ "loss": 0.543503713607788,
495
+ "step": 670
496
+ },
497
+ {
498
+ "epoch": 0.3589574397888486,
499
+ "grad_norm": 0.45412692427635193,
500
+ "learning_rate": 0.00013511111111111113,
501
+ "loss": 0.4648271560668945,
502
+ "step": 680
503
+ },
504
+ {
505
+ "epoch": 0.3642362256680963,
506
+ "grad_norm": 0.5843290686607361,
507
+ "learning_rate": 0.000134,
508
+ "loss": 0.4611194133758545,
509
+ "step": 690
510
+ },
511
+ {
512
+ "epoch": 0.3695150115473441,
513
+ "grad_norm": 0.5016586184501648,
514
+ "learning_rate": 0.00013288888888888888,
515
+ "loss": 0.5229296684265137,
516
+ "step": 700
517
+ },
518
+ {
519
+ "epoch": 0.37479379742659186,
520
+ "grad_norm": 0.594906747341156,
521
+ "learning_rate": 0.00013177777777777778,
522
+ "loss": 0.4897459506988525,
523
+ "step": 710
524
+ },
525
+ {
526
+ "epoch": 0.38007258330583965,
527
+ "grad_norm": 0.6894946098327637,
528
+ "learning_rate": 0.00013066666666666668,
529
+ "loss": 0.5067539691925049,
530
+ "step": 720
531
+ },
532
+ {
533
+ "epoch": 0.38535136918508744,
534
+ "grad_norm": 0.5899446606636047,
535
+ "learning_rate": 0.00012955555555555555,
536
+ "loss": 0.572650671005249,
537
+ "step": 730
538
+ },
539
+ {
540
+ "epoch": 0.3906301550643352,
541
+ "grad_norm": 0.5914052724838257,
542
+ "learning_rate": 0.00012844444444444446,
543
+ "loss": 0.49350743293762206,
544
+ "step": 740
545
+ },
546
+ {
547
+ "epoch": 0.395908940943583,
548
+ "grad_norm": 0.5701265931129456,
549
+ "learning_rate": 0.00012733333333333336,
550
+ "loss": 0.4886340141296387,
551
+ "step": 750
552
+ },
553
+ {
554
+ "epoch": 0.40118772682283077,
555
+ "grad_norm": 0.5554320216178894,
556
+ "learning_rate": 0.00012622222222222223,
557
+ "loss": 0.5152291774749755,
558
+ "step": 760
559
+ },
560
+ {
561
+ "epoch": 0.4064665127020785,
562
+ "grad_norm": 0.542721688747406,
563
+ "learning_rate": 0.0001251111111111111,
564
+ "loss": 0.5169697284698487,
565
+ "step": 770
566
+ },
567
+ {
568
+ "epoch": 0.4117452985813263,
569
+ "grad_norm": 0.5239601731300354,
570
+ "learning_rate": 0.000124,
571
+ "loss": 0.5226495265960693,
572
+ "step": 780
573
+ },
574
+ {
575
+ "epoch": 0.41702408446057404,
576
+ "grad_norm": 0.5628821849822998,
577
+ "learning_rate": 0.0001228888888888889,
578
+ "loss": 0.4952070236206055,
579
+ "step": 790
580
+ },
581
+ {
582
+ "epoch": 0.42230287033982183,
583
+ "grad_norm": 0.53690505027771,
584
+ "learning_rate": 0.0001217777777777778,
585
+ "loss": 0.5176570892333985,
586
+ "step": 800
587
+ },
588
+ {
589
+ "epoch": 0.42758165621906963,
590
+ "grad_norm": 0.47472912073135376,
591
+ "learning_rate": 0.00012066666666666668,
592
+ "loss": 0.48554182052612305,
593
+ "step": 810
594
+ },
595
+ {
596
+ "epoch": 0.43286044209831737,
597
+ "grad_norm": 0.5943832397460938,
598
+ "learning_rate": 0.00011955555555555556,
599
+ "loss": 0.5519014835357666,
600
+ "step": 820
601
+ },
602
+ {
603
+ "epoch": 0.43813922797756516,
604
+ "grad_norm": 0.5702099800109863,
605
+ "learning_rate": 0.00011844444444444444,
606
+ "loss": 0.5203097343444825,
607
+ "step": 830
608
+ },
609
+ {
610
+ "epoch": 0.44341801385681295,
611
+ "grad_norm": 0.5570870637893677,
612
+ "learning_rate": 0.00011733333333333334,
613
+ "loss": 0.5008646488189697,
614
+ "step": 840
615
+ },
616
+ {
617
+ "epoch": 0.4486967997360607,
618
+ "grad_norm": 0.6742717623710632,
619
+ "learning_rate": 0.00011622222222222223,
620
+ "loss": 0.4884360313415527,
621
+ "step": 850
622
+ },
623
+ {
624
+ "epoch": 0.4539755856153085,
625
+ "grad_norm": 0.5782440900802612,
626
+ "learning_rate": 0.00011511111111111112,
627
+ "loss": 0.497051477432251,
628
+ "step": 860
629
+ },
630
+ {
631
+ "epoch": 0.4592543714945563,
632
+ "grad_norm": 0.6157666444778442,
633
+ "learning_rate": 0.00011399999999999999,
634
+ "loss": 0.5011940956115722,
635
+ "step": 870
636
+ },
637
+ {
638
+ "epoch": 0.464533157373804,
639
+ "grad_norm": 0.4744451344013214,
640
+ "learning_rate": 0.0001128888888888889,
641
+ "loss": 0.4997218132019043,
642
+ "step": 880
643
+ },
644
+ {
645
+ "epoch": 0.4698119432530518,
646
+ "grad_norm": 0.5648385286331177,
647
+ "learning_rate": 0.00011177777777777778,
648
+ "loss": 0.5077646255493165,
649
+ "step": 890
650
+ },
651
+ {
652
+ "epoch": 0.47509072913229955,
653
+ "grad_norm": 0.5268929600715637,
654
+ "learning_rate": 0.00011066666666666667,
655
+ "loss": 0.4928537368774414,
656
+ "step": 900
657
+ },
658
+ {
659
+ "epoch": 0.48036951501154734,
660
+ "grad_norm": 0.5827873945236206,
661
+ "learning_rate": 0.00010955555555555557,
662
+ "loss": 0.5176413536071778,
663
+ "step": 910
664
+ },
665
+ {
666
+ "epoch": 0.48564830089079514,
667
+ "grad_norm": 0.5056174993515015,
668
+ "learning_rate": 0.00010844444444444446,
669
+ "loss": 0.5180400848388672,
670
+ "step": 920
671
+ },
672
+ {
673
+ "epoch": 0.4909270867700429,
674
+ "grad_norm": 0.47383710741996765,
675
+ "learning_rate": 0.00010733333333333333,
676
+ "loss": 0.4904231071472168,
677
+ "step": 930
678
+ },
679
+ {
680
+ "epoch": 0.49620587264929067,
681
+ "grad_norm": 0.5303201079368591,
682
+ "learning_rate": 0.00010622222222222222,
683
+ "loss": 0.4838510036468506,
684
+ "step": 940
685
+ },
686
+ {
687
+ "epoch": 0.5014846585285384,
688
+ "grad_norm": 0.6136945486068726,
689
+ "learning_rate": 0.00010511111111111112,
690
+ "loss": 0.4908186912536621,
691
+ "step": 950
692
+ },
693
+ {
694
+ "epoch": 0.5067634444077862,
695
+ "grad_norm": 0.6062966585159302,
696
+ "learning_rate": 0.00010400000000000001,
697
+ "loss": 0.5015266895294189,
698
+ "step": 960
699
+ },
700
+ {
701
+ "epoch": 0.512042230287034,
702
+ "grad_norm": 0.5623139142990112,
703
+ "learning_rate": 0.0001028888888888889,
704
+ "loss": 0.5161266803741456,
705
+ "step": 970
706
+ },
707
+ {
708
+ "epoch": 0.5173210161662818,
709
+ "grad_norm": 0.5548786520957947,
710
+ "learning_rate": 0.00010177777777777777,
711
+ "loss": 0.4842812538146973,
712
+ "step": 980
713
+ },
714
+ {
715
+ "epoch": 0.5225998020455296,
716
+ "grad_norm": 0.4839674234390259,
717
+ "learning_rate": 0.00010066666666666667,
718
+ "loss": 0.47931456565856934,
719
+ "step": 990
720
+ },
721
+ {
722
+ "epoch": 0.5278785879247773,
723
+ "grad_norm": 0.6106690764427185,
724
+ "learning_rate": 9.955555555555556e-05,
725
+ "loss": 0.5249813079833985,
726
+ "step": 1000
727
+ },
728
+ {
729
+ "epoch": 0.5278785879247773,
730
+ "eval_loss": 0.4978668689727783,
731
+ "eval_runtime": 459.5591,
732
+ "eval_samples_per_second": 24.373,
733
+ "eval_steps_per_second": 6.095,
734
+ "step": 1000
735
+ },
736
+ {
737
+ "epoch": 0.5278785879247773,
738
+ "step": 1000,
739
+ "test_best_wer": 0.21571711227233525,
740
+ "test_exact_match": 0.0,
741
+ "test_hallucination_rate": 0.0546875,
742
+ "test_wer": 0.3787225407133073
743
+ },
744
+ {
745
+ "epoch": 0.5331573738040251,
746
+ "grad_norm": 0.6384595036506653,
747
+ "learning_rate": 9.844444444444444e-05,
748
+ "loss": 0.4587059497833252,
749
+ "step": 1010
750
+ },
751
+ {
752
+ "epoch": 0.5384361596832729,
753
+ "grad_norm": 0.5651249289512634,
754
+ "learning_rate": 9.733333333333335e-05,
755
+ "loss": 0.4801759719848633,
756
+ "step": 1020
757
+ },
758
+ {
759
+ "epoch": 0.5437149455625206,
760
+ "grad_norm": 0.6035206913948059,
761
+ "learning_rate": 9.622222222222222e-05,
762
+ "loss": 0.581445837020874,
763
+ "step": 1030
764
+ },
765
+ {
766
+ "epoch": 0.5489937314417684,
767
+ "grad_norm": 0.6020093560218811,
768
+ "learning_rate": 9.511111111111112e-05,
769
+ "loss": 0.5043127536773682,
770
+ "step": 1040
771
+ },
772
+ {
773
+ "epoch": 0.5542725173210161,
774
+ "grad_norm": 0.5171729922294617,
775
+ "learning_rate": 9.4e-05,
776
+ "loss": 0.5017893791198731,
777
+ "step": 1050
778
+ },
779
+ {
780
+ "epoch": 0.5595513032002639,
781
+ "grad_norm": 0.4405026137828827,
782
+ "learning_rate": 9.28888888888889e-05,
783
+ "loss": 0.4623852252960205,
784
+ "step": 1060
785
+ },
786
+ {
787
+ "epoch": 0.5648300890795117,
788
+ "grad_norm": 0.5746331810951233,
789
+ "learning_rate": 9.177777777777778e-05,
790
+ "loss": 0.4706885814666748,
791
+ "step": 1070
792
+ },
793
+ {
794
+ "epoch": 0.5701088749587595,
795
+ "grad_norm": 0.5638077259063721,
796
+ "learning_rate": 9.066666666666667e-05,
797
+ "loss": 0.4803459644317627,
798
+ "step": 1080
799
+ },
800
+ {
801
+ "epoch": 0.5753876608380073,
802
+ "grad_norm": 0.5012305974960327,
803
+ "learning_rate": 8.955555555555556e-05,
804
+ "loss": 0.49541616439819336,
805
+ "step": 1090
806
+ },
807
+ {
808
+ "epoch": 0.580666446717255,
809
+ "grad_norm": 0.4939920902252197,
810
+ "learning_rate": 8.844444444444445e-05,
811
+ "loss": 0.5243307113647461,
812
+ "step": 1100
813
+ },
814
+ {
815
+ "epoch": 0.5859452325965028,
816
+ "grad_norm": 0.554180383682251,
817
+ "learning_rate": 8.733333333333333e-05,
818
+ "loss": 0.5223360538482666,
819
+ "step": 1110
820
+ },
821
+ {
822
+ "epoch": 0.5912240184757506,
823
+ "grad_norm": 0.6409974694252014,
824
+ "learning_rate": 8.622222222222222e-05,
825
+ "loss": 0.5112555503845215,
826
+ "step": 1120
827
+ },
828
+ {
829
+ "epoch": 0.5965028043549984,
830
+ "grad_norm": 0.6062772870063782,
831
+ "learning_rate": 8.511111111111112e-05,
832
+ "loss": 0.5176369190216065,
833
+ "step": 1130
834
+ },
835
+ {
836
+ "epoch": 0.6017815902342462,
837
+ "grad_norm": 0.5829088091850281,
838
+ "learning_rate": 8.4e-05,
839
+ "loss": 0.47949957847595215,
840
+ "step": 1140
841
+ },
842
+ {
843
+ "epoch": 0.607060376113494,
844
+ "grad_norm": 0.45387765765190125,
845
+ "learning_rate": 8.28888888888889e-05,
846
+ "loss": 0.48807411193847655,
847
+ "step": 1150
848
+ },
849
+ {
850
+ "epoch": 0.6123391619927416,
851
+ "grad_norm": 0.5486329793930054,
852
+ "learning_rate": 8.177777777777778e-05,
853
+ "loss": 0.5000582218170166,
854
+ "step": 1160
855
+ },
856
+ {
857
+ "epoch": 0.6176179478719894,
858
+ "grad_norm": 0.5097286701202393,
859
+ "learning_rate": 8.066666666666667e-05,
860
+ "loss": 0.5015737533569335,
861
+ "step": 1170
862
+ },
863
+ {
864
+ "epoch": 0.6228967337512372,
865
+ "grad_norm": 0.5109050869941711,
866
+ "learning_rate": 7.955555555555556e-05,
867
+ "loss": 0.5117743492126465,
868
+ "step": 1180
869
+ },
870
+ {
871
+ "epoch": 0.628175519630485,
872
+ "grad_norm": 0.5671198964118958,
873
+ "learning_rate": 7.844444444444446e-05,
874
+ "loss": 0.44252500534057615,
875
+ "step": 1190
876
+ },
877
+ {
878
+ "epoch": 0.6334543055097328,
879
+ "grad_norm": 0.6893628835678101,
880
+ "learning_rate": 7.733333333333333e-05,
881
+ "loss": 0.47486023902893065,
882
+ "step": 1200
883
+ },
884
+ {
885
+ "epoch": 0.6387330913889805,
886
+ "grad_norm": 0.5157479643821716,
887
+ "learning_rate": 7.622222222222223e-05,
888
+ "loss": 0.4672722816467285,
889
+ "step": 1210
890
+ },
891
+ {
892
+ "epoch": 0.6440118772682283,
893
+ "grad_norm": 0.6201571226119995,
894
+ "learning_rate": 7.511111111111111e-05,
895
+ "loss": 0.4837379455566406,
896
+ "step": 1220
897
+ },
898
+ {
899
+ "epoch": 0.6492906631474761,
900
+ "grad_norm": 0.5775447487831116,
901
+ "learning_rate": 7.4e-05,
902
+ "loss": 0.5213785171508789,
903
+ "step": 1230
904
+ },
905
+ {
906
+ "epoch": 0.6545694490267239,
907
+ "grad_norm": 0.4748264253139496,
908
+ "learning_rate": 7.28888888888889e-05,
909
+ "loss": 0.5322606086730957,
910
+ "step": 1240
911
+ },
912
+ {
913
+ "epoch": 0.6598482349059717,
914
+ "grad_norm": 0.45400553941726685,
915
+ "learning_rate": 7.177777777777777e-05,
916
+ "loss": 0.4156055450439453,
917
+ "step": 1250
918
+ },
919
+ {
920
+ "epoch": 0.6651270207852193,
921
+ "grad_norm": 0.48097214102745056,
922
+ "learning_rate": 7.066666666666667e-05,
923
+ "loss": 0.5314281463623047,
924
+ "step": 1260
925
+ },
926
+ {
927
+ "epoch": 0.6704058066644671,
928
+ "grad_norm": 0.6790676712989807,
929
+ "learning_rate": 6.955555555555556e-05,
930
+ "loss": 0.5225338935852051,
931
+ "step": 1270
932
+ },
933
+ {
934
+ "epoch": 0.6756845925437149,
935
+ "grad_norm": 0.5606982111930847,
936
+ "learning_rate": 6.844444444444445e-05,
937
+ "loss": 0.49805512428283694,
938
+ "step": 1280
939
+ },
940
+ {
941
+ "epoch": 0.6809633784229627,
942
+ "grad_norm": 0.5538866519927979,
943
+ "learning_rate": 6.733333333333333e-05,
944
+ "loss": 0.45378851890563965,
945
+ "step": 1290
946
+ },
947
+ {
948
+ "epoch": 0.6862421643022105,
949
+ "grad_norm": 0.5676766037940979,
950
+ "learning_rate": 6.622222222222224e-05,
951
+ "loss": 0.4546516895294189,
952
+ "step": 1300
953
+ },
954
+ {
955
+ "epoch": 0.6915209501814583,
956
+ "grad_norm": 0.528609573841095,
957
+ "learning_rate": 6.511111111111111e-05,
958
+ "loss": 0.5184505939483642,
959
+ "step": 1310
960
+ },
961
+ {
962
+ "epoch": 0.696799736060706,
963
+ "grad_norm": 0.5784614682197571,
964
+ "learning_rate": 6.400000000000001e-05,
965
+ "loss": 0.49085006713867185,
966
+ "step": 1320
967
+ },
968
+ {
969
+ "epoch": 0.7020785219399538,
970
+ "grad_norm": 0.6029508113861084,
971
+ "learning_rate": 6.28888888888889e-05,
972
+ "loss": 0.5110648155212403,
973
+ "step": 1330
974
+ },
975
+ {
976
+ "epoch": 0.7073573078192016,
977
+ "grad_norm": 0.7090053558349609,
978
+ "learning_rate": 6.177777777777779e-05,
979
+ "loss": 0.4798156261444092,
980
+ "step": 1340
981
+ },
982
+ {
983
+ "epoch": 0.7126360936984494,
984
+ "grad_norm": 0.6578977108001709,
985
+ "learning_rate": 6.066666666666667e-05,
986
+ "loss": 0.46619811058044436,
987
+ "step": 1350
988
+ },
989
+ {
990
+ "epoch": 0.7179148795776972,
991
+ "grad_norm": 0.5187491178512573,
992
+ "learning_rate": 5.9555555555555554e-05,
993
+ "loss": 0.4869066715240479,
994
+ "step": 1360
995
+ },
996
+ {
997
+ "epoch": 0.7231936654569449,
998
+ "grad_norm": 0.5641804337501526,
999
+ "learning_rate": 5.844444444444445e-05,
1000
+ "loss": 0.46739349365234373,
1001
+ "step": 1370
1002
+ },
1003
+ {
1004
+ "epoch": 0.7284724513361927,
1005
+ "grad_norm": 0.5477316975593567,
1006
+ "learning_rate": 5.7333333333333336e-05,
1007
+ "loss": 0.47148904800415037,
1008
+ "step": 1380
1009
+ },
1010
+ {
1011
+ "epoch": 0.7337512372154404,
1012
+ "grad_norm": 0.5800978541374207,
1013
+ "learning_rate": 5.622222222222222e-05,
1014
+ "loss": 0.45050225257873533,
1015
+ "step": 1390
1016
+ },
1017
+ {
1018
+ "epoch": 0.7390300230946882,
1019
+ "grad_norm": 0.4455048739910126,
1020
+ "learning_rate": 5.511111111111111e-05,
1021
+ "loss": 0.47357759475708006,
1022
+ "step": 1400
1023
+ },
1024
+ {
1025
+ "epoch": 0.744308808973936,
1026
+ "grad_norm": 0.44692400097846985,
1027
+ "learning_rate": 5.4000000000000005e-05,
1028
+ "loss": 0.4442988395690918,
1029
+ "step": 1410
1030
+ },
1031
+ {
1032
+ "epoch": 0.7495875948531837,
1033
+ "grad_norm": 0.5060186386108398,
1034
+ "learning_rate": 5.2888888888888885e-05,
1035
+ "loss": 0.4623199462890625,
1036
+ "step": 1420
1037
+ },
1038
+ {
1039
+ "epoch": 0.7548663807324315,
1040
+ "grad_norm": 0.5362889766693115,
1041
+ "learning_rate": 5.177777777777778e-05,
1042
+ "loss": 0.42566213607788084,
1043
+ "step": 1430
1044
+ },
1045
+ {
1046
+ "epoch": 0.7601451666116793,
1047
+ "grad_norm": 0.5910239219665527,
1048
+ "learning_rate": 5.0666666666666674e-05,
1049
+ "loss": 0.4922187328338623,
1050
+ "step": 1440
1051
+ },
1052
+ {
1053
+ "epoch": 0.7654239524909271,
1054
+ "grad_norm": 0.5109201073646545,
1055
+ "learning_rate": 4.955555555555556e-05,
1056
+ "loss": 0.4271512031555176,
1057
+ "step": 1450
1058
+ },
1059
+ {
1060
+ "epoch": 0.7707027383701749,
1061
+ "grad_norm": 0.6082547903060913,
1062
+ "learning_rate": 4.844444444444445e-05,
1063
+ "loss": 0.4993791103363037,
1064
+ "step": 1460
1065
+ },
1066
+ {
1067
+ "epoch": 0.7759815242494227,
1068
+ "grad_norm": 0.5516886115074158,
1069
+ "learning_rate": 4.7333333333333336e-05,
1070
+ "loss": 0.44429945945739746,
1071
+ "step": 1470
1072
+ },
1073
+ {
1074
+ "epoch": 0.7812603101286704,
1075
+ "grad_norm": 0.5361435413360596,
1076
+ "learning_rate": 4.6222222222222224e-05,
1077
+ "loss": 0.46475949287414553,
1078
+ "step": 1480
1079
+ },
1080
+ {
1081
+ "epoch": 0.7865390960079182,
1082
+ "grad_norm": 0.6853407025337219,
1083
+ "learning_rate": 4.511111111111112e-05,
1084
+ "loss": 0.48923401832580565,
1085
+ "step": 1490
1086
+ },
1087
+ {
1088
+ "epoch": 0.791817881887166,
1089
+ "grad_norm": 0.4870113432407379,
1090
+ "learning_rate": 4.4000000000000006e-05,
1091
+ "loss": 0.4502518653869629,
1092
+ "step": 1500
1093
+ },
1094
+ {
1095
+ "epoch": 0.791817881887166,
1096
+ "eval_loss": 0.4750659763813019,
1097
+ "eval_runtime": 460.037,
1098
+ "eval_samples_per_second": 24.348,
1099
+ "eval_steps_per_second": 6.089,
1100
+ "step": 1500
1101
+ }
1102
+ ],
1103
+ "logging_steps": 10,
1104
+ "max_steps": 1895,
1105
+ "num_input_tokens_seen": 0,
1106
+ "num_train_epochs": 1,
1107
+ "save_steps": 500,
1108
+ "stateful_callbacks": {
1109
+ "TrainerControl": {
1110
+ "args": {
1111
+ "should_epoch_stop": false,
1112
+ "should_evaluate": false,
1113
+ "should_log": false,
1114
+ "should_save": true,
1115
+ "should_training_stop": false
1116
+ },
1117
+ "attributes": {}
1118
+ }
1119
+ },
1120
+ "total_flos": 1.7942915216940672e+17,
1121
+ "train_batch_size": 2,
1122
+ "trial_name": null,
1123
+ "trial_params": null
1124
+ }
qwen3.5-2b/{checkpoint-100 → checkpoint-1500}/training_args.bin RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:292ecbf9094e5aa65f4e56a857f0e91efc5e8c116a774ff2556019858caa1768
3
- size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63f5227842d4c60d51e48d582480b968b2a26456e3111e78999104a00e72c35c
3
+ size 5265
qwen3.5-2b/tb/events.out.tfevents.1789942145.fed349279e2c.8211.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d8cb59132d4a5f5f32d6bf0c666a029e77792379d7374978e1a5e35c334efd38
3
- size 27752
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:45162d85cf935e854ba42f62cb22f0875bc40da0870da4cef5cee358b9b5fe26
3
+ size 38839
qwen3.5-2b/test_eval/metrics.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
- "wer": 0.3787225407133073,
3
  "exact_match": 0.0,
4
- "hallucination_rate": 0.0546875,
5
  "n_examples": 128
6
  }
 
1
  {
2
+ "wer": 0.5323423469128913,
3
  "exact_match": 0.0,
4
+ "hallucination_rate": 0.109375,
5
  "n_examples": 128
6
  }
qwen3.5-2b/test_eval/predictions.jsonl CHANGED
The diff for this file is too large to render. See raw diff