PedramR commited on
Commit
b499f59
·
verified ·
1 Parent(s): edc0fe6

checkpoint-300

Browse files
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0f335ac58ef7285c169a541a936deaa09de5a4bc337fc5a658163e8e56fecf5a
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb8b1146b9f6f2ab245c4a7915693bcdc7bc275ede46e0dcf0c25ffd99a8f47e
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fe247bb153f5e48a6347f7acd41b9583dc3dfb20f84f4de68a89222466c464bf
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d0bf3ae236b3fb3f1a313d130212811110b1aed4eeec3e4954922b0f1f59f836
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:6a9bb72a4281afdaa85d24b988cedaad2e7575e9088aa3ae95e536cd421d4b51
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4db15b331dc121cabbbdc8caf7b9d6dce32e716cfc423688abeb7d5224297e6f
3
  size 14645
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:97887b03ed5ddb9fb30ec5a630828d150a649222e1efaf740f45334b4842e9f4
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:edcf8aad81b9757527dfc94cf931298c7f5301b1b302bc98028445b6f5d61e52
3
  size 1465
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/trainer_state.json RENAMED
@@ -2,9 +2,9 @@
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
- "epoch": 1.4380403458213258,
6
  "eval_steps": 50,
7
- "global_step": 250,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -223,6 +223,49 @@
223
  "eval_samples_per_second": 56.413,
224
  "eval_steps_per_second": 14.103,
225
  "step": 250
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
226
  }
227
  ],
228
  "logging_steps": 10,
@@ -242,7 +285,7 @@
242
  "attributes": {}
243
  }
244
  },
245
- "total_flos": 1.6731342844263936e+16,
246
  "train_batch_size": 4,
247
  "trial_name": null,
248
  "trial_params": null
 
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 1.7262247838616713,
6
  "eval_steps": 50,
7
+ "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
223
  "eval_samples_per_second": 56.413,
224
  "eval_steps_per_second": 14.103,
225
  "step": 250
226
+ },
227
+ {
228
+ "epoch": 1.4956772334293948,
229
+ "grad_norm": 2.263134002685547,
230
+ "learning_rate": 5.313131313131313e-05,
231
+ "loss": 0.49338202476501464,
232
+ "step": 260
233
+ },
234
+ {
235
+ "epoch": 1.553314121037464,
236
+ "grad_norm": 3.2704994678497314,
237
+ "learning_rate": 5.111111111111111e-05,
238
+ "loss": 1.0119559288024902,
239
+ "step": 270
240
+ },
241
+ {
242
+ "epoch": 1.6109510086455332,
243
+ "grad_norm": 4.177541732788086,
244
+ "learning_rate": 4.909090909090909e-05,
245
+ "loss": 0.8234881401062012,
246
+ "step": 280
247
+ },
248
+ {
249
+ "epoch": 1.6685878962536023,
250
+ "grad_norm": 4.183651924133301,
251
+ "learning_rate": 4.7070707070707074e-05,
252
+ "loss": 0.6315438747406006,
253
+ "step": 290
254
+ },
255
+ {
256
+ "epoch": 1.7262247838616713,
257
+ "grad_norm": 3.0654070377349854,
258
+ "learning_rate": 4.5050505050505056e-05,
259
+ "loss": 0.6100435256958008,
260
+ "step": 300
261
+ },
262
+ {
263
+ "epoch": 1.7262247838616713,
264
+ "eval_loss": 0.9526227116584778,
265
+ "eval_runtime": 2.5488,
266
+ "eval_samples_per_second": 56.498,
267
+ "eval_steps_per_second": 14.124,
268
+ "step": 300
269
  }
270
  ],
271
  "logging_steps": 10,
 
285
  "attributes": {}
286
  }
287
  },
288
+ "total_flos": 1.9895526504301056e+16,
289
  "train_batch_size": 4,
290
  "trial_name": null,
291
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-250 → checkpoint-300}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b5312debb0485f92b7b7650e1f23a4dff3d3a6eac45b6c9317518cb14d07641a
3
- size 12001
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a120065397e08a6d467934d96155e37488f19b2b1f224a89b2c286d2d3200f00
3
+ size 13327