PedramR commited on
Commit
edc0fe6
·
verified ·
1 Parent(s): 7577bf2

checkpoint-250

Browse files
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7c7b2139bfa85954ecc4ddf56d29a80db597a7a2c1d04b75ae74e41688d0b1dc
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0f335ac58ef7285c169a541a936deaa09de5a4bc337fc5a658163e8e56fecf5a
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:04141f72da7ba24f697adc74fc80fee14be14daf03dc8f6325eeae461a3f9374
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fe247bb153f5e48a6347f7acd41b9583dc3dfb20f84f4de68a89222466c464bf
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:bcc015c243960e175ce6ad5a0955ff172eaeeb5aac901582b5cc34830cc6bc24
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6a9bb72a4281afdaa85d24b988cedaad2e7575e9088aa3ae95e536cd421d4b51
3
  size 14645
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4c2f4641c0edb2bfb241fbdc1196c117a5f1697dbfac8e5dc6f94774f897bc32
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:97887b03ed5ddb9fb30ec5a630828d150a649222e1efaf740f45334b4842e9f4
3
  size 1465
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/trainer_state.json RENAMED
@@ -2,9 +2,9 @@
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
- "epoch": 1.1498559077809798,
6
  "eval_steps": 50,
7
- "global_step": 200,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -180,6 +180,49 @@
180
  "eval_samples_per_second": 56.433,
181
  "eval_steps_per_second": 14.108,
182
  "step": 200
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
183
  }
184
  ],
185
  "logging_steps": 10,
@@ -199,7 +242,7 @@
199
  "attributes": {}
200
  }
201
  },
202
- "total_flos": 1.3413018767963136e+16,
203
  "train_batch_size": 4,
204
  "trial_name": null,
205
  "trial_params": null
 
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 1.4380403458213258,
6
  "eval_steps": 50,
7
+ "global_step": 250,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
180
  "eval_samples_per_second": 56.433,
181
  "eval_steps_per_second": 14.108,
182
  "step": 200
183
+ },
184
+ {
185
+ "epoch": 1.207492795389049,
186
+ "grad_norm": 2.812197208404541,
187
+ "learning_rate": 6.323232323232323e-05,
188
+ "loss": 0.5693333625793457,
189
+ "step": 210
190
+ },
191
+ {
192
+ "epoch": 1.2651296829971181,
193
+ "grad_norm": 3.991762161254883,
194
+ "learning_rate": 6.121212121212121e-05,
195
+ "loss": 0.7459357261657715,
196
+ "step": 220
197
+ },
198
+ {
199
+ "epoch": 1.3227665706051872,
200
+ "grad_norm": 2.750476360321045,
201
+ "learning_rate": 5.91919191919192e-05,
202
+ "loss": 0.8752019882202149,
203
+ "step": 230
204
+ },
205
+ {
206
+ "epoch": 1.3804034582132565,
207
+ "grad_norm": 3.3135452270507812,
208
+ "learning_rate": 5.717171717171717e-05,
209
+ "loss": 0.7731741428375244,
210
+ "step": 240
211
+ },
212
+ {
213
+ "epoch": 1.4380403458213258,
214
+ "grad_norm": 4.044683933258057,
215
+ "learning_rate": 5.5151515151515156e-05,
216
+ "loss": 0.6258707046508789,
217
+ "step": 250
218
+ },
219
+ {
220
+ "epoch": 1.4380403458213258,
221
+ "eval_loss": 0.9547940492630005,
222
+ "eval_runtime": 2.5526,
223
+ "eval_samples_per_second": 56.413,
224
+ "eval_steps_per_second": 14.103,
225
+ "step": 250
226
  }
227
  ],
228
  "logging_steps": 10,
 
242
  "attributes": {}
243
  }
244
  },
245
+ "total_flos": 1.6731342844263936e+16,
246
  "train_batch_size": 4,
247
  "trial_name": null,
248
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-200 → checkpoint-250}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4d9f66376a79084a22812581ddac3f00d23fc4c6c331bb0a0745865773d1563f
3
- size 10675
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b5312debb0485f92b7b7650e1f23a4dff3d3a6eac45b6c9317518cb14d07641a
3
+ size 12001