PedramR commited on
Commit
fa1177f
·
verified ·
1 Parent(s): 2deef5e

checkpoint-150

Browse files
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/README.md RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/adapter_config.json RENAMED
File without changes
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/adapter_model.safetensors RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:d832f0e7028771fce609568a9b4f180a858cb814f73d1e7249b02440397241f0
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c1728f9a704086e3589352b594b5e256dd7ff4e42ba8b23422f3b074bc59efce
3
  size 134604152
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/optimizer.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:3f35d00ee693c59692701624c37870bc1ac1fce5d3db9ccf3b386afab0acab59
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e1efdf3478d02050b220c46304cd72273ec6e592db8814768b70d6527ad63ce1
3
  size 269426431
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/rng_state.pth RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8106c15826296de5d71c00671c582d299e589cc9f7689d0cd37d905ba9ea5c6d
3
  size 14645
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:529f17025d0b7da5a0551fca62e4565a474ea67e2881c757146f59a354c567f9
3
  size 14645
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/scheduler.pt RENAMED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:466b26b2fb2e52bf171794d0d11032cdaddebdfd2afd0dcc0ab4252a3aa9acd2
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:78a8f80c807aa6a675c92ce4dd0be216c722daa1274f4adbc86cde14a2254e40
3
  size 1465
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/trainer_state.json RENAMED
@@ -1,10 +1,10 @@
1
  {
2
- "best_global_step": 50,
3
- "best_metric": 1.1613417863845825,
4
- "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-50",
5
- "epoch": 0.2881844380403458,
6
  "eval_steps": 50,
7
- "global_step": 50,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
@@ -51,6 +51,92 @@
51
  "eval_samples_per_second": 56.853,
52
  "eval_steps_per_second": 14.213,
53
  "step": 50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
54
  }
55
  ],
56
  "logging_steps": 10,
@@ -70,7 +156,7 @@
70
  "attributes": {}
71
  }
72
  },
73
- "total_flos": 3421478286965760.0,
74
  "train_batch_size": 4,
75
  "trial_name": null,
76
  "trial_params": null
 
1
  {
2
+ "best_global_step": 150,
3
+ "best_metric": 0.9355365037918091,
4
+ "best_model_checkpoint": "./outputs/edit-qwen3.5-2b/checkpoint-150",
5
+ "epoch": 0.8645533141210374,
6
  "eval_steps": 50,
7
+ "global_step": 150,
8
  "is_hyper_param_search": false,
9
  "is_local_process_zero": true,
10
  "is_world_process_zero": true,
 
51
  "eval_samples_per_second": 56.853,
52
  "eval_steps_per_second": 14.213,
53
  "step": 50
54
+ },
55
+ {
56
+ "epoch": 0.345821325648415,
57
+ "grad_norm": 3.6900596618652344,
58
+ "learning_rate": 9.353535353535354e-05,
59
+ "loss": 1.2613624572753905,
60
+ "step": 60
61
+ },
62
+ {
63
+ "epoch": 0.4034582132564842,
64
+ "grad_norm": 3.4724302291870117,
65
+ "learning_rate": 9.151515151515152e-05,
66
+ "loss": 0.9937849044799805,
67
+ "step": 70
68
+ },
69
+ {
70
+ "epoch": 0.4610951008645533,
71
+ "grad_norm": 3.44368052482605,
72
+ "learning_rate": 8.94949494949495e-05,
73
+ "loss": 1.165354347229004,
74
+ "step": 80
75
+ },
76
+ {
77
+ "epoch": 0.5187319884726225,
78
+ "grad_norm": 3.752511739730835,
79
+ "learning_rate": 8.747474747474747e-05,
80
+ "loss": 0.9866464614868165,
81
+ "step": 90
82
+ },
83
+ {
84
+ "epoch": 0.5763688760806917,
85
+ "grad_norm": 3.243701934814453,
86
+ "learning_rate": 8.545454545454545e-05,
87
+ "loss": 1.2611675262451172,
88
+ "step": 100
89
+ },
90
+ {
91
+ "epoch": 0.5763688760806917,
92
+ "eval_loss": 0.9719823002815247,
93
+ "eval_runtime": 2.5438,
94
+ "eval_samples_per_second": 56.607,
95
+ "eval_steps_per_second": 14.152,
96
+ "step": 100
97
+ },
98
+ {
99
+ "epoch": 0.6340057636887608,
100
+ "grad_norm": 5.817702293395996,
101
+ "learning_rate": 8.343434343434344e-05,
102
+ "loss": 1.1399730682373046,
103
+ "step": 110
104
+ },
105
+ {
106
+ "epoch": 0.69164265129683,
107
+ "grad_norm": 4.230010509490967,
108
+ "learning_rate": 8.141414141414141e-05,
109
+ "loss": 1.022247314453125,
110
+ "step": 120
111
+ },
112
+ {
113
+ "epoch": 0.7492795389048992,
114
+ "grad_norm": 5.1941914558410645,
115
+ "learning_rate": 7.93939393939394e-05,
116
+ "loss": 0.7813982009887696,
117
+ "step": 130
118
+ },
119
+ {
120
+ "epoch": 0.8069164265129684,
121
+ "grad_norm": 2.836448907852173,
122
+ "learning_rate": 7.737373737373738e-05,
123
+ "loss": 1.296474075317383,
124
+ "step": 140
125
+ },
126
+ {
127
+ "epoch": 0.8645533141210374,
128
+ "grad_norm": 4.288060188293457,
129
+ "learning_rate": 7.535353535353536e-05,
130
+ "loss": 1.1853026390075683,
131
+ "step": 150
132
+ },
133
+ {
134
+ "epoch": 0.8645533141210374,
135
+ "eval_loss": 0.9355365037918091,
136
+ "eval_runtime": 2.5492,
137
+ "eval_samples_per_second": 56.488,
138
+ "eval_steps_per_second": 14.122,
139
+ "step": 150
140
  }
141
  ],
142
  "logging_steps": 10,
 
156
  "attributes": {}
157
  }
158
  },
159
+ "total_flos": 1.0098881638347264e+16,
160
  "train_batch_size": 4,
161
  "trial_name": null,
162
  "trial_params": null
edit-qwen3.5-2b/{checkpoint-50 → checkpoint-150}/training_args.bin RENAMED
File without changes
edit-qwen3.5-2b/tb/events.out.tfevents.1790604388.9b7939721266.4215.0 CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:fa80a216c44b3ebde978198a5d88f07e7e3e839c246b6c2ba25e8ca1e5572d96
3
- size 8031
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff0b233bc0fa1c0b505edf088c3b3cf7dc519f375840228353d9368d4e3f697c
3
+ size 9349