PedramR commited on
Commit
8831ff0
·
verified ·
1 Parent(s): b499f59

checkpoint-350

Browse files
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:cb8b1146b9f6f2ab245c4a7915693bcdc7bc275ede46e0dcf0c25ffd99a8f47e
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:615027d4316ae38d0bd9a05f6f73a58a48f0a41299020c158d699dd2506e0333
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d0bf3ae236b3fb3f1a313d130212811110b1aed4eeec3e4954922b0f1f59f836
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e6a507dfe8a03d18a13d4e3ea82211fa095887561c48bfb931cbb41ae7118ee4
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:4db15b331dc121cabbbdc8caf7b9d6dce32e716cfc423688abeb7d5224297e6f
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb0cc07bf8584542fc3e445215bf5f344f42ca1fd0358903763df928113fd0ad
3
  size 14645
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:edcf8aad81b9757527dfc94cf931298c7f5301b1b302bc98028445b6f5d61e52
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06d22c6a60fc2beed1e26cbe201ba98f53ea7e3c9a4142c6b073b11a4a68afe7
3
  size 1465
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/trainer_state.json RENAMED
@@ -2,9 +2,9 @@
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
- "epoch": 1.7262247838616713,
6
  "eval_steps": 50,
7
- "global_step": 300,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -266,6 +266,49 @@
266
  "eval_samples_per_second": 56.498,
267
  "eval_steps_per_second": 14.124,
268
  "step": 300
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
269
  }
270
  ],
271
  "logging_steps": 10,
@@ -285,7 +328,7 @@
285
  "attributes": {}
286
  }
287
  },
288
- "total_flos": 1.9895526504301056e+16,
289
  "train_batch_size": 4,
290
  "trial_name": null,
291
  "trial_params": null
 
2
  "best_global_step": 150,
3
  "best_metric": 0.9355365037918091,
4
  "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 2.011527377521614,
6
  "eval_steps": 50,
7
+ "global_step": 350,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
266
  "eval_samples_per_second": 56.498,
267
  "eval_steps_per_second": 14.124,
268
  "step": 300
269
+ },
270
+ {
271
+ "epoch": 1.7838616714697406,
272
+ "grad_norm": 4.792970657348633,
273
+ "learning_rate": 4.303030303030303e-05,
274
+ "loss": 0.7996978282928466,
275
+ "step": 310
276
+ },
277
+ {
278
+ "epoch": 1.84149855907781,
279
+ "grad_norm": 3.7477433681488037,
280
+ "learning_rate": 4.101010101010101e-05,
281
+ "loss": 1.0061834335327149,
282
+ "step": 320
283
+ },
284
+ {
285
+ "epoch": 1.899135446685879,
286
+ "grad_norm": 5.449916839599609,
287
+ "learning_rate": 3.898989898989899e-05,
288
+ "loss": 0.702385139465332,
289
+ "step": 330
290
+ },
291
+ {
292
+ "epoch": 1.956772334293948,
293
+ "grad_norm": 4.3524322509765625,
294
+ "learning_rate": 3.6969696969696974e-05,
295
+ "loss": 0.7624853610992431,
296
+ "step": 340
297
+ },
298
+ {
299
+ "epoch": 2.011527377521614,
300
+ "grad_norm": 3.221466064453125,
301
+ "learning_rate": 3.494949494949495e-05,
302
+ "loss": 0.624992036819458,
303
+ "step": 350
304
+ },
305
+ {
306
+ "epoch": 2.011527377521614,
307
+ "eval_loss": 0.9362655282020569,
308
+ "eval_runtime": 2.554,
309
+ "eval_samples_per_second": 56.383,
310
+ "eval_steps_per_second": 14.096,
311
+ "step": 350
312
  }
313
  ],
314
  "logging_steps": 10,
 
328
  "attributes": {}
329
  }
330
  },
331
+ "total_flos": 2.3177856345391104e+16,
332
  "train_batch_size": 4,
333
  "trial_name": null,
334
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-300 → checkpoint-350}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:a120065397e08a6d467934d96155e37488f19b2b1f224a89b2c286d2d3200f00
3
- size 13327
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f9d7ad78e505fdfa48ca2010e6c24e3fa64a2b5f3c46e84ca6ef4021bab3e48b
3
+ size 14653