PedramR commited on
Commit
e243d7b
·
verified ·
1 Parent(s): 5665a3f

checkpoint-450

Browse files
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d0de40c4432cd3c1bf14a3f38149584b15af1ede17906f76f4dd71b1249755da
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:609465288589443a6141621f706d0e0a7a751a57dec19aa417f6718dabb20b6c
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b6fba941b664b2ee77f60d5cd2884c4d7d320d4ba1fd46aa7ebd3aa17ae7cb85
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:16bea85bbaa0c9c8182d7008601d4f963390a7785f4219c26dc827dae1d04e6d
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:480d7441c2e8f584c3db05ea6ab19542f9132f3782b5029d46b7c2f24752b3c6
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d1416146577c55503c3516767da9d2425f2969f6e0ff823b8f246492379b4756
3
  size 14645
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0d6935181675e1d1665ff39c91c33682acc867a658aded1287cdcaddd5b6d872
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f428fa345834e7d5a04466e8a86e74db282439f94f12e58c1904cbb0d1242e1a
3
  size 1465
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/trainer_state.json RENAMED
@@ -2,9 +2,9 @@
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
- "epoch": 2.2997118155619596,
6
  "eval_steps": 50,
7
- "global_step": 400,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -352,6 +352,49 @@
352
  "eval_samples_per_second": 56.419,
353
  "eval_steps_per_second": 14.105,
354
  "step": 400
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
355
  }
356
  ],
357
  "logging_steps": 10,
@@ -371,7 +414,7 @@
371
  "attributes": {}
372
  }
373
  },
374
- "total_flos": 2.6578399898932224e+16,
375
  "train_batch_size": 4,
376
  "trial_name": null,
377
  "trial_params": null
 
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 2.5878962536023056,
6
  "eval_steps": 50,
7
+ "global_step": 450,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
352
  "eval_samples_per_second": 56.419,
353
  "eval_steps_per_second": 14.105,
354
  "step": 400
355
+ },
356
+ {
357
+ "epoch": 2.357348703170029,
358
+ "grad_norm": 4.348433971405029,
359
+ "learning_rate": 2.282828282828283e-05,
360
+ "loss": 0.586982011795044,
361
+ "step": 410
362
+ },
363
+ {
364
+ "epoch": 2.414985590778098,
365
+ "grad_norm": 3.5055832862854004,
366
+ "learning_rate": 2.080808080808081e-05,
367
+ "loss": 0.39878134727478026,
368
+ "step": 420
369
+ },
370
+ {
371
+ "epoch": 2.472622478386167,
372
+ "grad_norm": 4.1734747886657715,
373
+ "learning_rate": 1.878787878787879e-05,
374
+ "loss": 0.48946046829223633,
375
+ "step": 430
376
+ },
377
+ {
378
+ "epoch": 2.5302593659942363,
379
+ "grad_norm": 3.2713208198547363,
380
+ "learning_rate": 1.6767676767676768e-05,
381
+ "loss": 0.4852273941040039,
382
+ "step": 440
383
+ },
384
+ {
385
+ "epoch": 2.5878962536023056,
386
+ "grad_norm": 3.623262882232666,
387
+ "learning_rate": 1.4747474747474749e-05,
388
+ "loss": 0.6178666591644287,
389
+ "step": 450
390
+ },
391
+ {
392
+ "epoch": 2.5878962536023056,
393
+ "eval_loss": 1.0052427053451538,
394
+ "eval_runtime": 2.5507,
395
+ "eval_samples_per_second": 56.456,
396
+ "eval_steps_per_second": 14.114,
397
+ "step": 450
398
  }
399
  ],
400
  "logging_steps": 10,
 
414
  "attributes": {}
415
  }
416
  },
417
+ "total_flos": 2.991792883425024e+16,
418
  "train_batch_size": 4,
419
  "trial_name": null,
420
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-400 → checkpoint-450}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:733499a8a9725ec95a1a377a6fd52119a63f9d808ed75b32cb36935fab6e4e4f
3
- size 15979
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e9823821ec0ad20dd1affaf3ee846d1a3da597075cfabf570f720d6992122504
3
+ size 17305