PedramR commited on
Commit
5665a3f
·
verified ·
1 Parent(s): 8831ff0

checkpoint-400

Browse files
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:615027d4316ae38d0bd9a05f6f73a58a48f0a41299020c158d699dd2506e0333
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d0de40c4432cd3c1bf14a3f38149584b15af1ede17906f76f4dd71b1249755da
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e6a507dfe8a03d18a13d4e3ea82211fa095887561c48bfb931cbb41ae7118ee4
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6fba941b664b2ee77f60d5cd2884c4d7d320d4ba1fd46aa7ebd3aa17ae7cb85
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fb0cc07bf8584542fc3e445215bf5f344f42ca1fd0358903763df928113fd0ad
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:480d7441c2e8f584c3db05ea6ab19542f9132f3782b5029d46b7c2f24752b3c6
3
  size 14645
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:06d22c6a60fc2beed1e26cbe201ba98f53ea7e3c9a4142c6b073b11a4a68afe7
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d6935181675e1d1665ff39c91c33682acc867a658aded1287cdcaddd5b6d872
3
  size 1465
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/trainer_state.json RENAMED
@@ -2,9 +2,9 @@
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
- "epoch": 2.011527377521614,
6
  "eval_steps": 50,
7
- "global_step": 350,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -309,6 +309,49 @@
309
  "eval_samples_per_second": 56.383,
310
  "eval_steps_per_second": 14.096,
311
  "step": 350
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
312
  }
313
  ],
314
  "logging_steps": 10,
@@ -328,7 +371,7 @@
328
  "attributes": {}
329
  }
330
  },
331
- "total_flos": 2.3177856345391104e+16,
332
  "train_batch_size": 4,
333
  "trial_name": null,
334
  "trial_params": null
 
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 2.2997118155619596,
6
  "eval_steps": 50,
7
+ "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
309
  "eval_samples_per_second": 56.383,
310
  "eval_steps_per_second": 14.096,
311
  "step": 350
312
+ },
313
+ {
314
+ "epoch": 2.069164265129683,
315
+ "grad_norm": 2.879030466079712,
316
+ "learning_rate": 3.292929292929293e-05,
317
+ "loss": 0.6604353904724121,
318
+ "step": 360
319
+ },
320
+ {
321
+ "epoch": 2.126801152737752,
322
+ "grad_norm": 3.0312411785125732,
323
+ "learning_rate": 3.090909090909091e-05,
324
+ "loss": 0.4823118209838867,
325
+ "step": 370
326
+ },
327
+ {
328
+ "epoch": 2.1844380403458215,
329
+ "grad_norm": 3.097743034362793,
330
+ "learning_rate": 2.8888888888888888e-05,
331
+ "loss": 0.4382351875305176,
332
+ "step": 380
333
+ },
334
+ {
335
+ "epoch": 2.2420749279538903,
336
+ "grad_norm": 2.786288261413574,
337
+ "learning_rate": 2.686868686868687e-05,
338
+ "loss": 0.3148329734802246,
339
+ "step": 390
340
+ },
341
+ {
342
+ "epoch": 2.2997118155619596,
343
+ "grad_norm": 3.8189518451690674,
344
+ "learning_rate": 2.4848484848484847e-05,
345
+ "loss": 0.7436184883117676,
346
+ "step": 400
347
+ },
348
+ {
349
+ "epoch": 2.2997118155619596,
350
+ "eval_loss": 1.0097650289535522,
351
+ "eval_runtime": 2.5523,
352
+ "eval_samples_per_second": 56.419,
353
+ "eval_steps_per_second": 14.105,
354
+ "step": 400
355
  }
356
  ],
357
  "logging_steps": 10,
 
371
  "attributes": {}
372
  }
373
  },
374
+ "total_flos": 2.6578399898932224e+16,
375
  "train_batch_size": 4,
376
  "trial_name": null,
377
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-350 → checkpoint-400}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:f9d7ad78e505fdfa48ca2010e6c24e3fa64a2b5f3c46e84ca6ef4021bab3e48b
3
- size 14653
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:733499a8a9725ec95a1a377a6fd52119a63f9d808ed75b32cb36935fab6e4e4f
3
+ size 15979