PedramR commited on
Commit
529fe08
·
verified ·
1 Parent(s): 9199b86

checkpoint-100

Browse files
qwen3.5-2b/best_checkpoint_wer/adapter_config.json CHANGED
@@ -32,17 +32,17 @@
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
35
- "v_proj",
36
- "up_proj",
37
- "in_proj_a",
38
  "gate_proj",
 
 
39
  "out_proj",
40
  "q_proj",
41
  "k_proj",
42
- "in_proj_b",
43
- "o_proj",
44
- "in_proj_qkv",
45
  "in_proj_z",
 
 
 
46
  "down_proj"
47
  ],
48
  "target_parameters": null,
 
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
35
+ "o_proj",
 
 
36
  "gate_proj",
37
+ "in_proj_qkv",
38
+ "in_proj_b",
39
  "out_proj",
40
  "q_proj",
41
  "k_proj",
 
 
 
42
  "in_proj_z",
43
+ "up_proj",
44
+ "v_proj",
45
+ "in_proj_a",
46
  "down_proj"
47
  ],
48
  "target_parameters": null,
qwen3.5-2b/best_checkpoint_wer/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b2e249ff273f8df09605bca89a1571361cd7bfc966c1a24f7672d253f488458f
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3516fb226a2b4bc6e564bc77e0f9bcf99ed02be7d1eb7622e36585ca32f548fe
3
  size 134604152
qwen3.5-2b/best_checkpoint_wer/best_metrics.json CHANGED
@@ -1,7 +1,7 @@
1
  {
2
  "step": 100,
3
- "wer": 0.46168322007318346,
4
- "exact_match": 0.00390625,
5
- "hallucination_rate": 0.09375,
6
- "n_examples": 256
7
  }
 
1
  {
2
  "step": 100,
3
+ "wer": 0.21571711227233525,
4
+ "exact_match": 0.0,
5
+ "hallucination_rate": 0.0,
6
+ "n_examples": 128
7
  }
qwen3.5-2b/best_checkpoint_wer/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:216255d2009f78676e6a71be2913f7485c161494a3b1b2ad6b608621d991684e
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fab5989da71e8335e77257636d2ba05a5e06e6c28915b4bcca735c064473e828
3
  size 269426431
qwen3.5-2b/best_checkpoint_wer/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5161ab58165f1ba32df42f8d11c180d6e456f2660b1a85bbaf6c2be2e076253a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:810230b50d30862ac3f8ee7cf9a4fed23d27151927c9bd74658f2abfc3b2d798
3
  size 1465
qwen3.5-2b/best_checkpoint_wer/trainer_state.json CHANGED
@@ -11,80 +11,80 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.007914523149980214,
14
- "grad_norm": 1.404574990272522,
15
- "learning_rate": 7.086614173228347e-06,
16
- "loss": 0.9118253707885742,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.015829046299960427,
21
- "grad_norm": 0.868704617023468,
22
- "learning_rate": 1.4960629921259845e-05,
23
- "loss": 0.8003530502319336,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.02374356944994064,
28
- "grad_norm": 0.8633992075920105,
29
- "learning_rate": 2.283464566929134e-05,
30
- "loss": 0.7220969676971436,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.031658092599920855,
35
- "grad_norm": 0.7782989740371704,
36
- "learning_rate": 3.070866141732284e-05,
37
- "loss": 0.7230673789978027,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.03957261574990107,
42
- "grad_norm": 1.0865058898925781,
43
- "learning_rate": 3.858267716535433e-05,
44
- "loss": 0.7225656986236573,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.04748713889988128,
49
- "grad_norm": 0.8686597943305969,
50
- "learning_rate": 4.645669291338583e-05,
51
- "loss": 0.6066381454467773,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.055401662049861494,
56
- "grad_norm": 0.9904829263687134,
57
- "learning_rate": 5.433070866141733e-05,
58
- "loss": 0.7127069473266602,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.06331618519984171,
63
- "grad_norm": 0.9868095517158508,
64
- "learning_rate": 6.220472440944882e-05,
65
- "loss": 0.6702294826507569,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.07123070834982193,
70
- "grad_norm": 1.1342442035675049,
71
- "learning_rate": 7.007874015748031e-05,
72
- "loss": 0.6693508148193359,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.07914523149980214,
77
- "grad_norm": 1.1382489204406738,
78
- "learning_rate": 7.795275590551181e-05,
79
- "loss": 0.6664952278137207,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.07914523149980214,
84
- "eval_loss": 0.6436748504638672,
85
- "eval_runtime": 461.1275,
86
- "eval_samples_per_second": 24.29,
87
- "eval_steps_per_second": 6.074,
88
  "step": 100
89
  }
90
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.007914523149980214,
14
+ "grad_norm": 1.111193060874939,
15
+ "learning_rate": 1.4173228346456694e-05,
16
+ "loss": 0.9034583091735839,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.015829046299960427,
21
+ "grad_norm": 0.8373973369598389,
22
+ "learning_rate": 2.992125984251969e-05,
23
+ "loss": 0.7779301166534424,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.02374356944994064,
28
+ "grad_norm": 0.7983554601669312,
29
+ "learning_rate": 4.566929133858268e-05,
30
+ "loss": 0.6945425987243652,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.031658092599920855,
35
+ "grad_norm": 0.7307385206222534,
36
+ "learning_rate": 6.141732283464568e-05,
37
+ "loss": 0.6987877368927002,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.03957261574990107,
42
+ "grad_norm": 1.0141053199768066,
43
+ "learning_rate": 7.716535433070867e-05,
44
+ "loss": 0.697762393951416,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.04748713889988128,
49
+ "grad_norm": 0.8023157715797424,
50
+ "learning_rate": 9.291338582677166e-05,
51
+ "loss": 0.5860283851623536,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.055401662049861494,
56
+ "grad_norm": 0.8767536282539368,
57
+ "learning_rate": 0.00010866141732283466,
58
+ "loss": 0.6935072422027588,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.06331618519984171,
63
+ "grad_norm": 0.7834345698356628,
64
+ "learning_rate": 0.00012440944881889765,
65
+ "loss": 0.6521744728088379,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.07123070834982193,
70
+ "grad_norm": 0.8145639896392822,
71
+ "learning_rate": 0.00014015748031496062,
72
+ "loss": 0.6522610664367676,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.07914523149980214,
77
+ "grad_norm": 0.9174225926399231,
78
+ "learning_rate": 0.00015590551181102362,
79
+ "loss": 0.6533907413482666,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.07914523149980214,
84
+ "eval_loss": 0.6276857256889343,
85
+ "eval_runtime": 460.4363,
86
+ "eval_samples_per_second": 24.327,
87
+ "eval_steps_per_second": 6.083,
88
  "step": 100
89
  }
90
  ],
qwen3.5-2b/best_checkpoint_wer/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:57ac5fc1677740bc79510bd3bb17e1d48a8cb3ceab532db738ff37384e2efc7b
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:292ecbf9094e5aa65f4e56a857f0e91efc5e8c116a774ff2556019858caa1768
3
  size 5201
qwen3.5-2b/checkpoint-100/adapter_config.json CHANGED
@@ -32,17 +32,17 @@
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
35
- "v_proj",
36
- "up_proj",
37
- "in_proj_a",
38
  "gate_proj",
 
 
39
  "out_proj",
40
  "q_proj",
41
  "k_proj",
42
- "in_proj_b",
43
- "o_proj",
44
- "in_proj_qkv",
45
  "in_proj_z",
 
 
 
46
  "down_proj"
47
  ],
48
  "target_parameters": null,
 
32
  "rank_pattern": {},
33
  "revision": null,
34
  "target_modules": [
35
+ "o_proj",
 
 
36
  "gate_proj",
37
+ "in_proj_qkv",
38
+ "in_proj_b",
39
  "out_proj",
40
  "q_proj",
41
  "k_proj",
 
 
 
42
  "in_proj_z",
43
+ "up_proj",
44
+ "v_proj",
45
+ "in_proj_a",
46
  "down_proj"
47
  ],
48
  "target_parameters": null,
qwen3.5-2b/checkpoint-100/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b2e249ff273f8df09605bca89a1571361cd7bfc966c1a24f7672d253f488458f
3
  size 134604152
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3516fb226a2b4bc6e564bc77e0f9bcf99ed02be7d1eb7622e36585ca32f548fe
3
  size 134604152
qwen3.5-2b/checkpoint-100/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:216255d2009f78676e6a71be2913f7485c161494a3b1b2ad6b608621d991684e
3
  size 269426431
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fab5989da71e8335e77257636d2ba05a5e06e6c28915b4bcca735c064473e828
3
  size 269426431
qwen3.5-2b/checkpoint-100/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:5161ab58165f1ba32df42f8d11c180d6e456f2660b1a85bbaf6c2be2e076253a
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:810230b50d30862ac3f8ee7cf9a4fed23d27151927c9bd74658f2abfc3b2d798
3
  size 1465
qwen3.5-2b/checkpoint-100/trainer_state.json CHANGED
@@ -11,80 +11,80 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.007914523149980214,
14
- "grad_norm": 1.404574990272522,
15
- "learning_rate": 7.086614173228347e-06,
16
- "loss": 0.9118253707885742,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.015829046299960427,
21
- "grad_norm": 0.868704617023468,
22
- "learning_rate": 1.4960629921259845e-05,
23
- "loss": 0.8003530502319336,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.02374356944994064,
28
- "grad_norm": 0.8633992075920105,
29
- "learning_rate": 2.283464566929134e-05,
30
- "loss": 0.7220969676971436,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.031658092599920855,
35
- "grad_norm": 0.7782989740371704,
36
- "learning_rate": 3.070866141732284e-05,
37
- "loss": 0.7230673789978027,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.03957261574990107,
42
- "grad_norm": 1.0865058898925781,
43
- "learning_rate": 3.858267716535433e-05,
44
- "loss": 0.7225656986236573,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.04748713889988128,
49
- "grad_norm": 0.8686597943305969,
50
- "learning_rate": 4.645669291338583e-05,
51
- "loss": 0.6066381454467773,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.055401662049861494,
56
- "grad_norm": 0.9904829263687134,
57
- "learning_rate": 5.433070866141733e-05,
58
- "loss": 0.7127069473266602,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.06331618519984171,
63
- "grad_norm": 0.9868095517158508,
64
- "learning_rate": 6.220472440944882e-05,
65
- "loss": 0.6702294826507569,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.07123070834982193,
70
- "grad_norm": 1.1342442035675049,
71
- "learning_rate": 7.007874015748031e-05,
72
- "loss": 0.6693508148193359,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.07914523149980214,
77
- "grad_norm": 1.1382489204406738,
78
- "learning_rate": 7.795275590551181e-05,
79
- "loss": 0.6664952278137207,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.07914523149980214,
84
- "eval_loss": 0.6436748504638672,
85
- "eval_runtime": 461.1275,
86
- "eval_samples_per_second": 24.29,
87
- "eval_steps_per_second": 6.074,
88
  "step": 100
89
  }
90
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.007914523149980214,
14
+ "grad_norm": 1.111193060874939,
15
+ "learning_rate": 1.4173228346456694e-05,
16
+ "loss": 0.9034583091735839,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.015829046299960427,
21
+ "grad_norm": 0.8373973369598389,
22
+ "learning_rate": 2.992125984251969e-05,
23
+ "loss": 0.7779301166534424,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.02374356944994064,
28
+ "grad_norm": 0.7983554601669312,
29
+ "learning_rate": 4.566929133858268e-05,
30
+ "loss": 0.6945425987243652,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.031658092599920855,
35
+ "grad_norm": 0.7307385206222534,
36
+ "learning_rate": 6.141732283464568e-05,
37
+ "loss": 0.6987877368927002,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.03957261574990107,
42
+ "grad_norm": 1.0141053199768066,
43
+ "learning_rate": 7.716535433070867e-05,
44
+ "loss": 0.697762393951416,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.04748713889988128,
49
+ "grad_norm": 0.8023157715797424,
50
+ "learning_rate": 9.291338582677166e-05,
51
+ "loss": 0.5860283851623536,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.055401662049861494,
56
+ "grad_norm": 0.8767536282539368,
57
+ "learning_rate": 0.00010866141732283466,
58
+ "loss": 0.6935072422027588,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.06331618519984171,
63
+ "grad_norm": 0.7834345698356628,
64
+ "learning_rate": 0.00012440944881889765,
65
+ "loss": 0.6521744728088379,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.07123070834982193,
70
+ "grad_norm": 0.8145639896392822,
71
+ "learning_rate": 0.00014015748031496062,
72
+ "loss": 0.6522610664367676,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.07914523149980214,
77
+ "grad_norm": 0.9174225926399231,
78
+ "learning_rate": 0.00015590551181102362,
79
+ "loss": 0.6533907413482666,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.07914523149980214,
84
+ "eval_loss": 0.6276857256889343,
85
+ "eval_runtime": 460.4363,
86
+ "eval_samples_per_second": 24.327,
87
+ "eval_steps_per_second": 6.083,
88
  "step": 100
89
  }
90
  ],
qwen3.5-2b/checkpoint-100/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:57ac5fc1677740bc79510bd3bb17e1d48a8cb3ceab532db738ff37384e2efc7b
3
  size 5201
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:292ecbf9094e5aa65f4e56a857f0e91efc5e8c116a774ff2556019858caa1768
3
  size 5201
qwen3.5-2b/config.yaml CHANGED
@@ -14,7 +14,7 @@ num_train_epochs: 2.0
14
  per_device_train_batch_size: 4
15
  per_device_eval_batch_size: 4
16
  gradient_accumulation_steps: 4
17
- learning_rate: 0.0001
18
  warmup_ratio: 0.05
19
  logging_steps: 10
20
  eval_steps: 100
@@ -36,8 +36,8 @@ test_split: test
36
  test_input_column: text_whisper
37
  test_target_column: text
38
  test_max_new_tokens: 1024
39
- test_batch_size: 32
40
- test_max_examples: 256
41
- test_checkpoint_max_examples: 256
42
  test_hallucination_overlap_floor: 50.0
43
  test_baseline: true
 
14
  per_device_train_batch_size: 4
15
  per_device_eval_batch_size: 4
16
  gradient_accumulation_steps: 4
17
+ learning_rate: 0.0002
18
  warmup_ratio: 0.05
19
  logging_steps: 10
20
  eval_steps: 100
 
36
  test_input_column: text_whisper
37
  test_target_column: text
38
  test_max_new_tokens: 1024
39
+ test_batch_size: 16
40
+ test_max_examples: 128
41
+ test_checkpoint_max_examples: 128
42
  test_hallucination_overlap_floor: 50.0
43
  test_baseline: true
qwen3.5-2b/resolved_config.yaml CHANGED
@@ -15,7 +15,7 @@ training:
15
  per_device_train_batch_size: 4
16
  per_device_eval_batch_size: 4
17
  gradient_accumulation_steps: 4
18
- learning_rate: 0.0001
19
  warmup_ratio: 0.05
20
  logging_steps: 10
21
  eval_steps: 100
@@ -41,7 +41,7 @@ test:
41
  input_column: text_whisper
42
  target_column: text
43
  max_new_tokens: 1024
44
- batch_size: 32
45
- max_examples: 256
46
- checkpoint_max_examples: 256
47
  resume_from_checkpoint: false
 
15
  per_device_train_batch_size: 4
16
  per_device_eval_batch_size: 4
17
  gradient_accumulation_steps: 4
18
+ learning_rate: 0.0002
19
  warmup_ratio: 0.05
20
  logging_steps: 10
21
  eval_steps: 100
 
41
  input_column: text_whisper
42
  target_column: text
43
  max_new_tokens: 1024
44
+ batch_size: 16
45
+ max_examples: 128
46
+ checkpoint_max_examples: 128
47
  resume_from_checkpoint: false
qwen3.5-2b/tb/events.out.tfevents.1789937702.fed349279e2c.7193.0 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9a9e21da477e32fbcfabdfb062c87b9aed9c9d40df1ee69822e367a6448d569a
3
+ size 8223
qwen3.5-2b/test_eval/metrics.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
- "wer": 0.46168322007318346,
3
- "exact_match": 0.00390625,
4
- "hallucination_rate": 0.09375,
5
- "n_examples": 256
6
  }
 
1
  {
2
+ "wer": 0.21571711227233525,
3
+ "exact_match": 0.0,
4
+ "hallucination_rate": 0.0,
5
+ "n_examples": 128
6
  }
qwen3.5-2b/test_eval/predictions.jsonl CHANGED
The diff for this file is too large to render. See raw diff
 
qwen3.5-2b/test_eval_baseline/metrics.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
- "wer": 0.27462624150548876,
3
  "exact_match": 0.0,
4
  "hallucination_rate": 0.015625,
5
- "n_examples": 256
6
  }
 
1
  {
2
+ "wer": 0.28090913703008474,
3
  "exact_match": 0.0,
4
  "hallucination_rate": 0.015625,
5
+ "n_examples": 128
6
  }
qwen3.5-2b/test_eval_baseline/predictions.jsonl CHANGED
The diff for this file is too large to render. See raw diff